loupe / evaluate_generalisation.py
anxuanng's picture
Loupe research preview: audit code, read weights, benchmark reports
84c3565 verified
Raw History Blame Contribute Delete
16 kB
"""How well the page audit holds on live public sites it was never tuned on.
Three questions, each answered with a number:
1. Detection: how many repeated groups the class-free detector finds per site, against the old same-class one.
2. Pixel checks against the browser: for every check read from pixels, the browser's own ratio gives the true
verdict against the default brief. Reported as false flags and misses, without tolerance and with a tolerance
calibrated on half of the sites and tested on the other half.
3. Geometry checks by fault injection: one known fault is applied to a live page (an item shifted down, pushed
sideways, made shorter; a heading indented) and the audit must flag the matching check where the untouched
page did not. The flag rate on untouched pages is reported next to it.
The "fresh" set was collected after every threshold, tolerance and scale had been fixed on the first set
(24 home pages). Nothing is calibrated on it: it only reports. Its contrast reference is not the declared
colours but the page rendered twice, with and without text.
Usage: evaluate_generalisation.py collect|report [first|fresh]
"""
import json
import numpy
import subprocess
import sys
from concurrent.futures import ThreadPoolExecutor
from pathlib import Path
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
PACKAGE = Path(__file__).resolve().parent
sys.path.insert(0, str(PACKAGE))
OUTPUT = PACKAGE / "generalisation"
SITES = {
"python": "https://www.python.org/", "rust": "https://www.rust-lang.org/", "django": "https://www.djangoproject.com/",
"bootstrap": "https://getbootstrap.com/", "tailwind": "https://tailwindcss.com/", "vue": "https://vuejs.org/",
"svelte": "https://svelte.dev/", "node": "https://nodejs.org/en", "mozilla": "https://www.mozilla.org/en-US/",
"govuk": "https://www.gov.uk/", "wikipedia": "https://en.wikipedia.org/wiki/Typography", "mdn": "https://developer.mozilla.org/en-US/",
"github": "https://github.com/features", "stripe": "https://stripe.com/", "shopify": "https://www.shopify.com/",
"ikea": "https://www.ikea.com/us/en/", "bbc": "https://www.bbc.com/", "guardian": "https://www.theguardian.com/international",
"nasa": "https://www.nasa.gov/", "mit": "https://www.mit.edu/", "figma": "https://www.figma.com/",
"notion": "https://www.notion.com/", "vercel": "https://vercel.com/", "astro": "https://astro.build/",
}
FRESH = {
"apple": ["https://www.apple.com/", "https://www.apple.com/mac/", "https://www.apple.com/iphone/"],
"ruby": ["https://www.ruby-lang.org/en/", "https://www.ruby-lang.org/en/documentation/", "https://www.ruby-lang.org/en/downloads/"],
"go": ["https://go.dev/", "https://go.dev/learn/", "https://go.dev/solutions/"],
"postgres": ["https://www.postgresql.org/", "https://www.postgresql.org/about/", "https://www.postgresql.org/download/"],
"kotlin": ["https://kotlinlang.org/", "https://kotlinlang.org/docs/home.html", "https://kotlinlang.org/community/"],
"typescript": ["https://www.typescriptlang.org/", "https://www.typescriptlang.org/docs/", "https://www.typescriptlang.org/download/"],
"w3c": ["https://www.w3.org/", "https://www.w3.org/standards/", "https://www.w3.org/about/"],
"npr": ["https://www.npr.org/", "https://www.npr.org/sections/news/", "https://www.npr.org/sections/culture/"],
"allbirds": ["https://www.allbirds.com/", "https://www.allbirds.com/collections/mens", "https://www.allbirds.com/pages/our-story"],
"harvard": ["https://www.harvard.edu/", "https://www.harvard.edu/about/", "https://www.harvard.edu/academics/"],
"docker": ["https://www.docker.com/", "https://www.docker.com/products/docker-desktop/", "https://www.docker.com/pricing/"],
"slack": ["https://slack.com/", "https://slack.com/features", "https://slack.com/pricing"],
}
THIRD = {
"php": ["https://www.php.net/", "https://www.php.net/docs.php", "https://www.php.net/downloads"],
"elixir": ["https://elixir-lang.org/", "https://elixir-lang.org/learning.html", "https://elixir-lang.org/install.html"],
"raspberrypi": ["https://www.raspberrypi.com/", "https://www.raspberrypi.com/products/", "https://www.raspberrypi.com/software/"],
"mongodb": ["https://www.mongodb.com/", "https://www.mongodb.com/pricing", "https://www.mongodb.com/company"],
"dropbox": ["https://www.dropbox.com/", "https://www.dropbox.com/features", "https://www.dropbox.com/plans"],
"stanford": ["https://www.stanford.edu/", "https://www.stanford.edu/about/", "https://www.stanford.edu/academics/"],
"who": ["https://www.who.int/", "https://www.who.int/about", "https://www.who.int/health-topics"],
"aljazeera": ["https://www.aljazeera.com/", "https://www.aljazeera.com/news/", "https://www.aljazeera.com/sports/"],
"gitlab": ["https://about.gitlab.com/", "https://about.gitlab.com/pricing/", "https://about.gitlab.com/platform/"],
"cloudflare": ["https://www.cloudflare.com/", "https://www.cloudflare.com/plans/", "https://www.cloudflare.com/learning/"],
"zendesk": ["https://www.zendesk.com/", "https://www.zendesk.com/pricing/", "https://www.zendesk.com/service/"],
"muji": ["https://www.muji.us/", "https://www.muji.us/collections/stationery", "https://www.muji.us/pages/about-muji"],
}
pages_of = lambda sites: {f"{site}-{index + 1}": url for site, urls in sites.items() for index, url in enumerate(urls)}
SETS = {"first": (SITES, PACKAGE / "generalisation"),
# The fresh pages collected again once line boxes existed: the set the line-box reading was developed on.
"dev": (pages_of(FRESH), PACKAGE / "generalisation_dev"),
# Pages first collected after the line-box reading was fixed: its unseen test.
"third": (pages_of(THIRD), PACKAGE / "generalisation_third"),
"fresh": ({f"{site}-{index + 1}": url for site, urls in FRESH.items() for index, url in enumerate(urls)}, PACKAGE / "generalisation_fresh")}
FAULTS = {"shift": "row_alignment", "gap": "row_gap_evenness", "height": "item_height_evenness", "indent": "text_alignment"}
PIXEL_CHECKS = ("type_scale", "line_spacing", "line_length", "text_contrast")
GEOMETRY_CHECKS = ("text_alignment", "row_alignment", "row_gap_evenness", "item_height_evenness", "item_width_evenness",
"image_shape_evenness", "column_alignment", "column_gap_evenness")
TOLERANCE_PERCENTILE = 80
def collect_one(job):
name, url, variant = job
directory = OUTPUT / name / variant
if (directory / "layout.json").exists():
return name, variant, "kept"
arguments = ["node", str(PACKAGE / "collect_page.cjs"), url, str(directory), "1440", "truth"]
if variant != "clean":
arguments.append(variant)
try:
completed = subprocess.run(arguments, capture_output=True, text=True, encoding="utf-8", timeout=180)
except subprocess.TimeoutExpired:
return name, variant, "timeout"
return name, variant, "ok" if completed.returncode == 0 else completed.stderr.strip().splitlines()[-1][:160]
def collect(variants=("clean", *FAULTS)):
jobs = [(name, url, variant) for name, url in SITES.items() for variant in variants]
with ThreadPoolExecutor(max_workers=4) as pool:
for name, variant, status in pool.map(collect_one, jobs):
print(name, variant, status, flush=True)
def rendered_contrast(pixels, without_text, box):
"""Contrast of a text block from the page rendered with and without its text; None when the two differ too little or too much."""
import block_measure
left, top, width, height = (int(round(value)) for value in box)
painted = pixels[max(top, 0):top + height, max(left, 0):left + width].astype(numpy.int32)
behind = without_text[max(top, 0):top + height, max(left, 0):left + width].astype(numpy.int32)
if painted.shape != behind.shape or painted.size == 0:
return None
difference = numpy.abs(painted - behind).sum(axis=2)
glyphs = difference > 24
# Under a dozen changed pixels is no text; most of the box changing means the page moved between the two pictures.
if glyphs.sum() < 12 or glyphs.mean() > 0.7:
return None
core = difference >= numpy.percentile(difference[glyphs], 90)
return block_measure.contrast_ratio(behind[core].mean(axis=0), painted[core].mean(axis=0))
def true_ratio(result, layout, pictures=None):
"""The same ratio from the browser's computed values (or, for contrast, from the two renderings), or None."""
import validate_page_measures
import block_measure
truths = [layout["texts"][text_id]["truth"] for text_id in result["text_ids"]]
if result["check"] == "type_scale":
return truths[0]["font_px"] / truths[1]["font_px"]
if result["check"] == "line_spacing":
return truths[0]["line_height_px"] / truths[0]["font_px"] if truths[0]["line_height_px"] else None
if result["check"] == "line_length":
return result["numerator"]["value"] / truths[0]["font_px"]
if pictures is not None:
return rendered_contrast(pictures[0], pictures[1], layout["texts"][result["text_ids"][0]]["box"])
behind = validate_page_measures.colour(truths[0]["background"])
return block_measure.contrast_ratio(behind, validate_page_measures.colour(truths[0]["color"], behind))
def verdict(ratio, band, tolerance=0.0):
if ratio < band[0] / (1 + tolerance):
return "too low"
return "too high" if ratio > band[1] * (1 + tolerance) else "in range"
def pixel_rows(name, results, layout, pictures=None):
rows = []
for result in results:
if result["check"] not in PIXEL_CHECKS:
continue
truth = true_ratio(result, layout, pictures if result["check"] == "text_contrast" else None)
if truth:
rows.append({"site": name, "check": result["check"], "ratio": result["ratio"], "truth": truth, "band": result["band"]})
return rows
def agreement(rows, tolerances):
table = {}
for row in rows:
entry = table.setdefault(row["check"], {"checks": 0, "truly_out": 0, "false_flags": 0, "misses": 0, "wrong_side": 0})
read = verdict(row["ratio"], row["band"], tolerances.get(row["check"], 0.0))
real = verdict(row["truth"], row["band"])
entry["checks"] += 1
entry["truly_out"] += real != "in range"
entry["false_flags"] += read != "in range" and real == "in range"
entry["misses"] += read == "in range" and real != "in range"
entry["wrong_side"] += "in range" not in (read, real) and read != real
for entry in table.values():
truly_in = entry["checks"] - entry["truly_out"]
entry["false_flag_rate"] = entry["false_flags"] / truly_in if truly_in else None
entry["miss_rate"] = entry["misses"] / entry["truly_out"] if entry["truly_out"] else None
return table
def flagged(results, kind):
return sum(result["check"] == kind and result["advice"] != "keep" for result in results)
def report():
import numpy
import page_audit
read = page_audit.load_read()
audits = {}
layouts = {}
for name in SITES:
for variant in ("clean", *FAULTS):
directory = OUTPUT / name / variant
if not (directory / "layout.json").exists():
continue
layouts[name, variant] = json.loads((directory / "layout.json").read_text(encoding="utf-8"))
audits[name, variant] = page_audit.judge(page_audit.measure(directory), page_audit.DEFAULT_BRIEF, read)
clean_sites = [name for name in SITES if (name, "clean") in audits]
detection = {name: {"texts": len(layouts[name, "clean"]["texts"]),
"same_class_card_groups": len(layouts[name, "clean"]["card_groups"]),
"repeated_groups": len(layouts[name, "clean"]["repeats"]),
"checks": len(audits[name, "clean"])} for name in clean_sites}
calibration_sites, test_sites = clean_sites[0::2], clean_sites[1::2]
rows = [row for name in clean_sites for row in pixel_rows(name, audits[name, "clean"], layouts[name, "clean"])]
errors = {}
for row in rows:
if row["site"] in calibration_sites:
errors.setdefault(row["check"], []).append(abs(row["ratio"] / row["truth"] - 1))
tolerances = {kind: float(numpy.percentile(values, TOLERANCE_PERCENTILE)) for kind, values in errors.items()}
# Contrast is read exactly where the text sits on a flat colour and is far off elsewhere (text over pictures),
# so a tolerance cannot describe it; it keeps none.
tolerances["text_contrast"] = 0.0
test_rows = [row for row in rows if row["site"] in test_sites]
relative_error = {kind: {"checks": len(values), "median": float(numpy.median(values)), "percentile_80": float(numpy.percentile(values, 80))}
for kind in PIXEL_CHECKS
for values in [[abs(row["ratio"] / row["truth"] - 1) for row in rows if row["check"] == kind]] if values}
injection = {}
for fault, kind in FAULTS.items():
entry = injection.setdefault(fault, {"check": kind, "sites_with_a_target": 0, "caught": 0, "missed_on": []})
for name in clean_sites:
if not layouts.get((name, fault), {}).get("fault_applied"):
continue
entry["sites_with_a_target"] += 1
if flagged(audits[name, fault], kind) > flagged(audits[name, "clean"], kind):
entry["caught"] += 1
else:
entry["missed_on"].append(name)
untouched = {}
for kind in GEOMETRY_CHECKS:
results = [result for name in clean_sites for result in audits[name, "clean"] if result["check"] == kind]
if results:
untouched[kind] = {"checks": len(results), "flagged": sum(result["advice"] != "keep" for result in results),
"flag_rate": sum(result["advice"] != "keep" for result in results) / len(results)}
summary = {"sites_collected": clean_sites, "sites_failed": [name for name in SITES if name not in clean_sites],
"detection": detection, "calibration_sites": calibration_sites, "test_sites": test_sites,
"tolerance_percentile": TOLERANCE_PERCENTILE, "tolerance": tolerances, "pixel_relative_error": relative_error,
"pixel_verdicts_on_test_sites": {"without_tolerance": agreement(test_rows, {}), "with_tolerance": agreement(test_rows, tolerances)},
"fault_injection": injection, "geometry_flags_on_untouched_sites": untouched}
(OUTPUT / "report.json").write_text(json.dumps(summary, indent=1), encoding="utf-8")
(PACKAGE / "tolerances.json").write_text(json.dumps({"source": "generalisation/report.json", "percentile": TOLERANCE_PERCENTILE,
"tolerance": tolerances}, indent=1), encoding="utf-8")
print("sites", len(clean_sites), "failed", summary["sites_failed"])
print("groups: same-class", sum(entry["same_class_card_groups"] for entry in detection.values()),
"class-free", sum(entry["repeated_groups"] for entry in detection.values()),
"| sites with any:", sum(entry["same_class_card_groups"] > 0 for entry in detection.values()),
"->", sum(entry["repeated_groups"] > 0 for entry in detection.values()))
print("tolerance", {kind: round(value, 3) for kind, value in tolerances.items()})
print("relative error", {kind: (entry["checks"], round(entry["median"], 3), round(entry["percentile_80"], 3)) for kind, entry in relative_error.items()})
for label, table in summary["pixel_verdicts_on_test_sites"].items():
print(label)
for kind, entry in table.items():
print(" ", kind, entry)
for fault, entry in injection.items():
print("fault", fault, entry)
for kind, entry in untouched.items():
print("untouched", kind, entry["checks"], entry["flagged"], round(entry["flag_rate"], 3))
if __name__ == "__main__":
collect() if sys.argv[1] == "collect" else report()