Download evaluate_generalisation.py from codepawl/loupe: direct link, hf CLI and curl.
- Browser
- Download file 16 kB
-
https://huggingface.co/codepawl/loupe/resolve/main/evaluate_generalisation.py
- Command line
-
hf download hf://codepawl/loupe/evaluate_generalisation.py
-
curl -L -o evaluate_generalisation.py https://huggingface.co/codepawl/loupe/resolve/main/evaluate_generalisation.py
16 kB
| """How well the page audit holds on live public sites it was never tuned on. | |
| Three questions, each answered with a number: | |
| 1. Detection: how many repeated groups the class-free detector finds per site, against the old same-class one. | |
| 2. Pixel checks against the browser: for every check read from pixels, the browser's own ratio gives the true | |
| verdict against the default brief. Reported as false flags and misses, without tolerance and with a tolerance | |
| calibrated on half of the sites and tested on the other half. | |
| 3. Geometry checks by fault injection: one known fault is applied to a live page (an item shifted down, pushed | |
| sideways, made shorter; a heading indented) and the audit must flag the matching check where the untouched | |
| page did not. The flag rate on untouched pages is reported next to it. | |
| The "fresh" set was collected after every threshold, tolerance and scale had been fixed on the first set | |
| (24 home pages). Nothing is calibrated on it: it only reports. Its contrast reference is not the declared | |
| colours but the page rendered twice, with and without text. | |
| Usage: evaluate_generalisation.py collect|report [first|fresh] | |
| """ | |
| import json | |
| import numpy | |
| import subprocess | |
| import sys | |
| from concurrent.futures import ThreadPoolExecutor | |
| from pathlib import Path | |
| sys.stdout.reconfigure(encoding="utf-8", errors="replace") | |
| PACKAGE = Path(__file__).resolve().parent | |
| sys.path.insert(0, str(PACKAGE)) | |
| OUTPUT = PACKAGE / "generalisation" | |
| SITES = { | |
| "python": "https://www.python.org/", "rust": "https://www.rust-lang.org/", "django": "https://www.djangoproject.com/", | |
| "bootstrap": "https://getbootstrap.com/", "tailwind": "https://tailwindcss.com/", "vue": "https://vuejs.org/", | |
| "svelte": "https://svelte.dev/", "node": "https://nodejs.org/en", "mozilla": "https://www.mozilla.org/en-US/", | |
| "govuk": "https://www.gov.uk/", "wikipedia": "https://en.wikipedia.org/wiki/Typography", "mdn": "https://developer.mozilla.org/en-US/", | |
| "github": "https://github.com/features", "stripe": "https://stripe.com/", "shopify": "https://www.shopify.com/", | |
| "ikea": "https://www.ikea.com/us/en/", "bbc": "https://www.bbc.com/", "guardian": "https://www.theguardian.com/international", | |
| "nasa": "https://www.nasa.gov/", "mit": "https://www.mit.edu/", "figma": "https://www.figma.com/", | |
| "notion": "https://www.notion.com/", "vercel": "https://vercel.com/", "astro": "https://astro.build/", | |
| } | |
| FRESH = { | |
| "apple": ["https://www.apple.com/", "https://www.apple.com/mac/", "https://www.apple.com/iphone/"], | |
| "ruby": ["https://www.ruby-lang.org/en/", "https://www.ruby-lang.org/en/documentation/", "https://www.ruby-lang.org/en/downloads/"], | |
| "go": ["https://go.dev/", "https://go.dev/learn/", "https://go.dev/solutions/"], | |
| "postgres": ["https://www.postgresql.org/", "https://www.postgresql.org/about/", "https://www.postgresql.org/download/"], | |
| "kotlin": ["https://kotlinlang.org/", "https://kotlinlang.org/docs/home.html", "https://kotlinlang.org/community/"], | |
| "typescript": ["https://www.typescriptlang.org/", "https://www.typescriptlang.org/docs/", "https://www.typescriptlang.org/download/"], | |
| "w3c": ["https://www.w3.org/", "https://www.w3.org/standards/", "https://www.w3.org/about/"], | |
| "npr": ["https://www.npr.org/", "https://www.npr.org/sections/news/", "https://www.npr.org/sections/culture/"], | |
| "allbirds": ["https://www.allbirds.com/", "https://www.allbirds.com/collections/mens", "https://www.allbirds.com/pages/our-story"], | |
| "harvard": ["https://www.harvard.edu/", "https://www.harvard.edu/about/", "https://www.harvard.edu/academics/"], | |
| "docker": ["https://www.docker.com/", "https://www.docker.com/products/docker-desktop/", "https://www.docker.com/pricing/"], | |
| "slack": ["https://slack.com/", "https://slack.com/features", "https://slack.com/pricing"], | |
| } | |
| THIRD = { | |
| "php": ["https://www.php.net/", "https://www.php.net/docs.php", "https://www.php.net/downloads"], | |
| "elixir": ["https://elixir-lang.org/", "https://elixir-lang.org/learning.html", "https://elixir-lang.org/install.html"], | |
| "raspberrypi": ["https://www.raspberrypi.com/", "https://www.raspberrypi.com/products/", "https://www.raspberrypi.com/software/"], | |
| "mongodb": ["https://www.mongodb.com/", "https://www.mongodb.com/pricing", "https://www.mongodb.com/company"], | |
| "dropbox": ["https://www.dropbox.com/", "https://www.dropbox.com/features", "https://www.dropbox.com/plans"], | |
| "stanford": ["https://www.stanford.edu/", "https://www.stanford.edu/about/", "https://www.stanford.edu/academics/"], | |
| "who": ["https://www.who.int/", "https://www.who.int/about", "https://www.who.int/health-topics"], | |
| "aljazeera": ["https://www.aljazeera.com/", "https://www.aljazeera.com/news/", "https://www.aljazeera.com/sports/"], | |
| "gitlab": ["https://about.gitlab.com/", "https://about.gitlab.com/pricing/", "https://about.gitlab.com/platform/"], | |
| "cloudflare": ["https://www.cloudflare.com/", "https://www.cloudflare.com/plans/", "https://www.cloudflare.com/learning/"], | |
| "zendesk": ["https://www.zendesk.com/", "https://www.zendesk.com/pricing/", "https://www.zendesk.com/service/"], | |
| "muji": ["https://www.muji.us/", "https://www.muji.us/collections/stationery", "https://www.muji.us/pages/about-muji"], | |
| } | |
| pages_of = lambda sites: {f"{site}-{index + 1}": url for site, urls in sites.items() for index, url in enumerate(urls)} | |
| SETS = {"first": (SITES, PACKAGE / "generalisation"), | |
| # The fresh pages collected again once line boxes existed: the set the line-box reading was developed on. | |
| "dev": (pages_of(FRESH), PACKAGE / "generalisation_dev"), | |
| # Pages first collected after the line-box reading was fixed: its unseen test. | |
| "third": (pages_of(THIRD), PACKAGE / "generalisation_third"), | |
| "fresh": ({f"{site}-{index + 1}": url for site, urls in FRESH.items() for index, url in enumerate(urls)}, PACKAGE / "generalisation_fresh")} | |
| FAULTS = {"shift": "row_alignment", "gap": "row_gap_evenness", "height": "item_height_evenness", "indent": "text_alignment"} | |
| PIXEL_CHECKS = ("type_scale", "line_spacing", "line_length", "text_contrast") | |
| GEOMETRY_CHECKS = ("text_alignment", "row_alignment", "row_gap_evenness", "item_height_evenness", "item_width_evenness", | |
| "image_shape_evenness", "column_alignment", "column_gap_evenness") | |
| TOLERANCE_PERCENTILE = 80 | |
| def collect_one(job): | |
| name, url, variant = job | |
| directory = OUTPUT / name / variant | |
| if (directory / "layout.json").exists(): | |
| return name, variant, "kept" | |
| arguments = ["node", str(PACKAGE / "collect_page.cjs"), url, str(directory), "1440", "truth"] | |
| if variant != "clean": | |
| arguments.append(variant) | |
| try: | |
| completed = subprocess.run(arguments, capture_output=True, text=True, encoding="utf-8", timeout=180) | |
| except subprocess.TimeoutExpired: | |
| return name, variant, "timeout" | |
| return name, variant, "ok" if completed.returncode == 0 else completed.stderr.strip().splitlines()[-1][:160] | |
| def collect(variants=("clean", *FAULTS)): | |
| jobs = [(name, url, variant) for name, url in SITES.items() for variant in variants] | |
| with ThreadPoolExecutor(max_workers=4) as pool: | |
| for name, variant, status in pool.map(collect_one, jobs): | |
| print(name, variant, status, flush=True) | |
| def rendered_contrast(pixels, without_text, box): | |
| """Contrast of a text block from the page rendered with and without its text; None when the two differ too little or too much.""" | |
| import block_measure | |
| left, top, width, height = (int(round(value)) for value in box) | |
| painted = pixels[max(top, 0):top + height, max(left, 0):left + width].astype(numpy.int32) | |
| behind = without_text[max(top, 0):top + height, max(left, 0):left + width].astype(numpy.int32) | |
| if painted.shape != behind.shape or painted.size == 0: | |
| return None | |
| difference = numpy.abs(painted - behind).sum(axis=2) | |
| glyphs = difference > 24 | |
| # Under a dozen changed pixels is no text; most of the box changing means the page moved between the two pictures. | |
| if glyphs.sum() < 12 or glyphs.mean() > 0.7: | |
| return None | |
| core = difference >= numpy.percentile(difference[glyphs], 90) | |
| return block_measure.contrast_ratio(behind[core].mean(axis=0), painted[core].mean(axis=0)) | |
| def true_ratio(result, layout, pictures=None): | |
| """The same ratio from the browser's computed values (or, for contrast, from the two renderings), or None.""" | |
| import validate_page_measures | |
| import block_measure | |
| truths = [layout["texts"][text_id]["truth"] for text_id in result["text_ids"]] | |
| if result["check"] == "type_scale": | |
| return truths[0]["font_px"] / truths[1]["font_px"] | |
| if result["check"] == "line_spacing": | |
| return truths[0]["line_height_px"] / truths[0]["font_px"] if truths[0]["line_height_px"] else None | |
| if result["check"] == "line_length": | |
| return result["numerator"]["value"] / truths[0]["font_px"] | |
| if pictures is not None: | |
| return rendered_contrast(pictures[0], pictures[1], layout["texts"][result["text_ids"][0]]["box"]) | |
| behind = validate_page_measures.colour(truths[0]["background"]) | |
| return block_measure.contrast_ratio(behind, validate_page_measures.colour(truths[0]["color"], behind)) | |
| def verdict(ratio, band, tolerance=0.0): | |
| if ratio < band[0] / (1 + tolerance): | |
| return "too low" | |
| return "too high" if ratio > band[1] * (1 + tolerance) else "in range" | |
| def pixel_rows(name, results, layout, pictures=None): | |
| rows = [] | |
| for result in results: | |
| if result["check"] not in PIXEL_CHECKS: | |
| continue | |
| truth = true_ratio(result, layout, pictures if result["check"] == "text_contrast" else None) | |
| if truth: | |
| rows.append({"site": name, "check": result["check"], "ratio": result["ratio"], "truth": truth, "band": result["band"]}) | |
| return rows | |
| def agreement(rows, tolerances): | |
| table = {} | |
| for row in rows: | |
| entry = table.setdefault(row["check"], {"checks": 0, "truly_out": 0, "false_flags": 0, "misses": 0, "wrong_side": 0}) | |
| read = verdict(row["ratio"], row["band"], tolerances.get(row["check"], 0.0)) | |
| real = verdict(row["truth"], row["band"]) | |
| entry["checks"] += 1 | |
| entry["truly_out"] += real != "in range" | |
| entry["false_flags"] += read != "in range" and real == "in range" | |
| entry["misses"] += read == "in range" and real != "in range" | |
| entry["wrong_side"] += "in range" not in (read, real) and read != real | |
| for entry in table.values(): | |
| truly_in = entry["checks"] - entry["truly_out"] | |
| entry["false_flag_rate"] = entry["false_flags"] / truly_in if truly_in else None | |
| entry["miss_rate"] = entry["misses"] / entry["truly_out"] if entry["truly_out"] else None | |
| return table | |
| def flagged(results, kind): | |
| return sum(result["check"] == kind and result["advice"] != "keep" for result in results) | |
| def report(): | |
| import numpy | |
| import page_audit | |
| read = page_audit.load_read() | |
| audits = {} | |
| layouts = {} | |
| for name in SITES: | |
| for variant in ("clean", *FAULTS): | |
| directory = OUTPUT / name / variant | |
| if not (directory / "layout.json").exists(): | |
| continue | |
| layouts[name, variant] = json.loads((directory / "layout.json").read_text(encoding="utf-8")) | |
| audits[name, variant] = page_audit.judge(page_audit.measure(directory), page_audit.DEFAULT_BRIEF, read) | |
| clean_sites = [name for name in SITES if (name, "clean") in audits] | |
| detection = {name: {"texts": len(layouts[name, "clean"]["texts"]), | |
| "same_class_card_groups": len(layouts[name, "clean"]["card_groups"]), | |
| "repeated_groups": len(layouts[name, "clean"]["repeats"]), | |
| "checks": len(audits[name, "clean"])} for name in clean_sites} | |
| calibration_sites, test_sites = clean_sites[0::2], clean_sites[1::2] | |
| rows = [row for name in clean_sites for row in pixel_rows(name, audits[name, "clean"], layouts[name, "clean"])] | |
| errors = {} | |
| for row in rows: | |
| if row["site"] in calibration_sites: | |
| errors.setdefault(row["check"], []).append(abs(row["ratio"] / row["truth"] - 1)) | |
| tolerances = {kind: float(numpy.percentile(values, TOLERANCE_PERCENTILE)) for kind, values in errors.items()} | |
| # Contrast is read exactly where the text sits on a flat colour and is far off elsewhere (text over pictures), | |
| # so a tolerance cannot describe it; it keeps none. | |
| tolerances["text_contrast"] = 0.0 | |
| test_rows = [row for row in rows if row["site"] in test_sites] | |
| relative_error = {kind: {"checks": len(values), "median": float(numpy.median(values)), "percentile_80": float(numpy.percentile(values, 80))} | |
| for kind in PIXEL_CHECKS | |
| for values in [[abs(row["ratio"] / row["truth"] - 1) for row in rows if row["check"] == kind]] if values} | |
| injection = {} | |
| for fault, kind in FAULTS.items(): | |
| entry = injection.setdefault(fault, {"check": kind, "sites_with_a_target": 0, "caught": 0, "missed_on": []}) | |
| for name in clean_sites: | |
| if not layouts.get((name, fault), {}).get("fault_applied"): | |
| continue | |
| entry["sites_with_a_target"] += 1 | |
| if flagged(audits[name, fault], kind) > flagged(audits[name, "clean"], kind): | |
| entry["caught"] += 1 | |
| else: | |
| entry["missed_on"].append(name) | |
| untouched = {} | |
| for kind in GEOMETRY_CHECKS: | |
| results = [result for name in clean_sites for result in audits[name, "clean"] if result["check"] == kind] | |
| if results: | |
| untouched[kind] = {"checks": len(results), "flagged": sum(result["advice"] != "keep" for result in results), | |
| "flag_rate": sum(result["advice"] != "keep" for result in results) / len(results)} | |
| summary = {"sites_collected": clean_sites, "sites_failed": [name for name in SITES if name not in clean_sites], | |
| "detection": detection, "calibration_sites": calibration_sites, "test_sites": test_sites, | |
| "tolerance_percentile": TOLERANCE_PERCENTILE, "tolerance": tolerances, "pixel_relative_error": relative_error, | |
| "pixel_verdicts_on_test_sites": {"without_tolerance": agreement(test_rows, {}), "with_tolerance": agreement(test_rows, tolerances)}, | |
| "fault_injection": injection, "geometry_flags_on_untouched_sites": untouched} | |
| (OUTPUT / "report.json").write_text(json.dumps(summary, indent=1), encoding="utf-8") | |
| (PACKAGE / "tolerances.json").write_text(json.dumps({"source": "generalisation/report.json", "percentile": TOLERANCE_PERCENTILE, | |
| "tolerance": tolerances}, indent=1), encoding="utf-8") | |
| print("sites", len(clean_sites), "failed", summary["sites_failed"]) | |
| print("groups: same-class", sum(entry["same_class_card_groups"] for entry in detection.values()), | |
| "class-free", sum(entry["repeated_groups"] for entry in detection.values()), | |
| "| sites with any:", sum(entry["same_class_card_groups"] > 0 for entry in detection.values()), | |
| "->", sum(entry["repeated_groups"] > 0 for entry in detection.values())) | |
| print("tolerance", {kind: round(value, 3) for kind, value in tolerances.items()}) | |
| print("relative error", {kind: (entry["checks"], round(entry["median"], 3), round(entry["percentile_80"], 3)) for kind, entry in relative_error.items()}) | |
| for label, table in summary["pixel_verdicts_on_test_sites"].items(): | |
| print(label) | |
| for kind, entry in table.items(): | |
| print(" ", kind, entry) | |
| for fault, entry in injection.items(): | |
| print("fault", fault, entry) | |
| for kind, entry in untouched.items(): | |
| print("untouched", kind, entry["checks"], entry["flagged"], round(entry["flag_rate"], 3)) | |
| if __name__ == "__main__": | |
| collect() if sys.argv[1] == "collect" else report() | |