File size: 16,027 Bytes
84c3565 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 | """How well the page audit holds on live public sites it was never tuned on.
Three questions, each answered with a number:
1. Detection: how many repeated groups the class-free detector finds per site, against the old same-class one.
2. Pixel checks against the browser: for every check read from pixels, the browser's own ratio gives the true
verdict against the default brief. Reported as false flags and misses, without tolerance and with a tolerance
calibrated on half of the sites and tested on the other half.
3. Geometry checks by fault injection: one known fault is applied to a live page (an item shifted down, pushed
sideways, made shorter; a heading indented) and the audit must flag the matching check where the untouched
page did not. The flag rate on untouched pages is reported next to it.
The "fresh" set was collected after every threshold, tolerance and scale had been fixed on the first set
(24 home pages). Nothing is calibrated on it: it only reports. Its contrast reference is not the declared
colours but the page rendered twice, with and without text.
Usage: evaluate_generalisation.py collect|report [first|fresh]
"""
import json
import numpy
import subprocess
import sys
from concurrent.futures import ThreadPoolExecutor
from pathlib import Path
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
PACKAGE = Path(__file__).resolve().parent
sys.path.insert(0, str(PACKAGE))
OUTPUT = PACKAGE / "generalisation"
SITES = {
"python": "https://www.python.org/", "rust": "https://www.rust-lang.org/", "django": "https://www.djangoproject.com/",
"bootstrap": "https://getbootstrap.com/", "tailwind": "https://tailwindcss.com/", "vue": "https://vuejs.org/",
"svelte": "https://svelte.dev/", "node": "https://nodejs.org/en", "mozilla": "https://www.mozilla.org/en-US/",
"govuk": "https://www.gov.uk/", "wikipedia": "https://en.wikipedia.org/wiki/Typography", "mdn": "https://developer.mozilla.org/en-US/",
"github": "https://github.com/features", "stripe": "https://stripe.com/", "shopify": "https://www.shopify.com/",
"ikea": "https://www.ikea.com/us/en/", "bbc": "https://www.bbc.com/", "guardian": "https://www.theguardian.com/international",
"nasa": "https://www.nasa.gov/", "mit": "https://www.mit.edu/", "figma": "https://www.figma.com/",
"notion": "https://www.notion.com/", "vercel": "https://vercel.com/", "astro": "https://astro.build/",
}
FRESH = {
"apple": ["https://www.apple.com/", "https://www.apple.com/mac/", "https://www.apple.com/iphone/"],
"ruby": ["https://www.ruby-lang.org/en/", "https://www.ruby-lang.org/en/documentation/", "https://www.ruby-lang.org/en/downloads/"],
"go": ["https://go.dev/", "https://go.dev/learn/", "https://go.dev/solutions/"],
"postgres": ["https://www.postgresql.org/", "https://www.postgresql.org/about/", "https://www.postgresql.org/download/"],
"kotlin": ["https://kotlinlang.org/", "https://kotlinlang.org/docs/home.html", "https://kotlinlang.org/community/"],
"typescript": ["https://www.typescriptlang.org/", "https://www.typescriptlang.org/docs/", "https://www.typescriptlang.org/download/"],
"w3c": ["https://www.w3.org/", "https://www.w3.org/standards/", "https://www.w3.org/about/"],
"npr": ["https://www.npr.org/", "https://www.npr.org/sections/news/", "https://www.npr.org/sections/culture/"],
"allbirds": ["https://www.allbirds.com/", "https://www.allbirds.com/collections/mens", "https://www.allbirds.com/pages/our-story"],
"harvard": ["https://www.harvard.edu/", "https://www.harvard.edu/about/", "https://www.harvard.edu/academics/"],
"docker": ["https://www.docker.com/", "https://www.docker.com/products/docker-desktop/", "https://www.docker.com/pricing/"],
"slack": ["https://slack.com/", "https://slack.com/features", "https://slack.com/pricing"],
}
THIRD = {
"php": ["https://www.php.net/", "https://www.php.net/docs.php", "https://www.php.net/downloads"],
"elixir": ["https://elixir-lang.org/", "https://elixir-lang.org/learning.html", "https://elixir-lang.org/install.html"],
"raspberrypi": ["https://www.raspberrypi.com/", "https://www.raspberrypi.com/products/", "https://www.raspberrypi.com/software/"],
"mongodb": ["https://www.mongodb.com/", "https://www.mongodb.com/pricing", "https://www.mongodb.com/company"],
"dropbox": ["https://www.dropbox.com/", "https://www.dropbox.com/features", "https://www.dropbox.com/plans"],
"stanford": ["https://www.stanford.edu/", "https://www.stanford.edu/about/", "https://www.stanford.edu/academics/"],
"who": ["https://www.who.int/", "https://www.who.int/about", "https://www.who.int/health-topics"],
"aljazeera": ["https://www.aljazeera.com/", "https://www.aljazeera.com/news/", "https://www.aljazeera.com/sports/"],
"gitlab": ["https://about.gitlab.com/", "https://about.gitlab.com/pricing/", "https://about.gitlab.com/platform/"],
"cloudflare": ["https://www.cloudflare.com/", "https://www.cloudflare.com/plans/", "https://www.cloudflare.com/learning/"],
"zendesk": ["https://www.zendesk.com/", "https://www.zendesk.com/pricing/", "https://www.zendesk.com/service/"],
"muji": ["https://www.muji.us/", "https://www.muji.us/collections/stationery", "https://www.muji.us/pages/about-muji"],
}
pages_of = lambda sites: {f"{site}-{index + 1}": url for site, urls in sites.items() for index, url in enumerate(urls)}
SETS = {"first": (SITES, PACKAGE / "generalisation"),
# The fresh pages collected again once line boxes existed: the set the line-box reading was developed on.
"dev": (pages_of(FRESH), PACKAGE / "generalisation_dev"),
# Pages first collected after the line-box reading was fixed: its unseen test.
"third": (pages_of(THIRD), PACKAGE / "generalisation_third"),
"fresh": ({f"{site}-{index + 1}": url for site, urls in FRESH.items() for index, url in enumerate(urls)}, PACKAGE / "generalisation_fresh")}
FAULTS = {"shift": "row_alignment", "gap": "row_gap_evenness", "height": "item_height_evenness", "indent": "text_alignment"}
PIXEL_CHECKS = ("type_scale", "line_spacing", "line_length", "text_contrast")
GEOMETRY_CHECKS = ("text_alignment", "row_alignment", "row_gap_evenness", "item_height_evenness", "item_width_evenness",
"image_shape_evenness", "column_alignment", "column_gap_evenness")
TOLERANCE_PERCENTILE = 80
def collect_one(job):
name, url, variant = job
directory = OUTPUT / name / variant
if (directory / "layout.json").exists():
return name, variant, "kept"
arguments = ["node", str(PACKAGE / "collect_page.cjs"), url, str(directory), "1440", "truth"]
if variant != "clean":
arguments.append(variant)
try:
completed = subprocess.run(arguments, capture_output=True, text=True, encoding="utf-8", timeout=180)
except subprocess.TimeoutExpired:
return name, variant, "timeout"
return name, variant, "ok" if completed.returncode == 0 else completed.stderr.strip().splitlines()[-1][:160]
def collect(variants=("clean", *FAULTS)):
jobs = [(name, url, variant) for name, url in SITES.items() for variant in variants]
with ThreadPoolExecutor(max_workers=4) as pool:
for name, variant, status in pool.map(collect_one, jobs):
print(name, variant, status, flush=True)
def rendered_contrast(pixels, without_text, box):
"""Contrast of a text block from the page rendered with and without its text; None when the two differ too little or too much."""
import block_measure
left, top, width, height = (int(round(value)) for value in box)
painted = pixels[max(top, 0):top + height, max(left, 0):left + width].astype(numpy.int32)
behind = without_text[max(top, 0):top + height, max(left, 0):left + width].astype(numpy.int32)
if painted.shape != behind.shape or painted.size == 0:
return None
difference = numpy.abs(painted - behind).sum(axis=2)
glyphs = difference > 24
# Under a dozen changed pixels is no text; most of the box changing means the page moved between the two pictures.
if glyphs.sum() < 12 or glyphs.mean() > 0.7:
return None
core = difference >= numpy.percentile(difference[glyphs], 90)
return block_measure.contrast_ratio(behind[core].mean(axis=0), painted[core].mean(axis=0))
def true_ratio(result, layout, pictures=None):
"""The same ratio from the browser's computed values (or, for contrast, from the two renderings), or None."""
import validate_page_measures
import block_measure
truths = [layout["texts"][text_id]["truth"] for text_id in result["text_ids"]]
if result["check"] == "type_scale":
return truths[0]["font_px"] / truths[1]["font_px"]
if result["check"] == "line_spacing":
return truths[0]["line_height_px"] / truths[0]["font_px"] if truths[0]["line_height_px"] else None
if result["check"] == "line_length":
return result["numerator"]["value"] / truths[0]["font_px"]
if pictures is not None:
return rendered_contrast(pictures[0], pictures[1], layout["texts"][result["text_ids"][0]]["box"])
behind = validate_page_measures.colour(truths[0]["background"])
return block_measure.contrast_ratio(behind, validate_page_measures.colour(truths[0]["color"], behind))
def verdict(ratio, band, tolerance=0.0):
if ratio < band[0] / (1 + tolerance):
return "too low"
return "too high" if ratio > band[1] * (1 + tolerance) else "in range"
def pixel_rows(name, results, layout, pictures=None):
rows = []
for result in results:
if result["check"] not in PIXEL_CHECKS:
continue
truth = true_ratio(result, layout, pictures if result["check"] == "text_contrast" else None)
if truth:
rows.append({"site": name, "check": result["check"], "ratio": result["ratio"], "truth": truth, "band": result["band"]})
return rows
def agreement(rows, tolerances):
table = {}
for row in rows:
entry = table.setdefault(row["check"], {"checks": 0, "truly_out": 0, "false_flags": 0, "misses": 0, "wrong_side": 0})
read = verdict(row["ratio"], row["band"], tolerances.get(row["check"], 0.0))
real = verdict(row["truth"], row["band"])
entry["checks"] += 1
entry["truly_out"] += real != "in range"
entry["false_flags"] += read != "in range" and real == "in range"
entry["misses"] += read == "in range" and real != "in range"
entry["wrong_side"] += "in range" not in (read, real) and read != real
for entry in table.values():
truly_in = entry["checks"] - entry["truly_out"]
entry["false_flag_rate"] = entry["false_flags"] / truly_in if truly_in else None
entry["miss_rate"] = entry["misses"] / entry["truly_out"] if entry["truly_out"] else None
return table
def flagged(results, kind):
return sum(result["check"] == kind and result["advice"] != "keep" for result in results)
def report():
import numpy
import page_audit
read = page_audit.load_read()
audits = {}
layouts = {}
for name in SITES:
for variant in ("clean", *FAULTS):
directory = OUTPUT / name / variant
if not (directory / "layout.json").exists():
continue
layouts[name, variant] = json.loads((directory / "layout.json").read_text(encoding="utf-8"))
audits[name, variant] = page_audit.judge(page_audit.measure(directory), page_audit.DEFAULT_BRIEF, read)
clean_sites = [name for name in SITES if (name, "clean") in audits]
detection = {name: {"texts": len(layouts[name, "clean"]["texts"]),
"same_class_card_groups": len(layouts[name, "clean"]["card_groups"]),
"repeated_groups": len(layouts[name, "clean"]["repeats"]),
"checks": len(audits[name, "clean"])} for name in clean_sites}
calibration_sites, test_sites = clean_sites[0::2], clean_sites[1::2]
rows = [row for name in clean_sites for row in pixel_rows(name, audits[name, "clean"], layouts[name, "clean"])]
errors = {}
for row in rows:
if row["site"] in calibration_sites:
errors.setdefault(row["check"], []).append(abs(row["ratio"] / row["truth"] - 1))
tolerances = {kind: float(numpy.percentile(values, TOLERANCE_PERCENTILE)) for kind, values in errors.items()}
# Contrast is read exactly where the text sits on a flat colour and is far off elsewhere (text over pictures),
# so a tolerance cannot describe it; it keeps none.
tolerances["text_contrast"] = 0.0
test_rows = [row for row in rows if row["site"] in test_sites]
relative_error = {kind: {"checks": len(values), "median": float(numpy.median(values)), "percentile_80": float(numpy.percentile(values, 80))}
for kind in PIXEL_CHECKS
for values in [[abs(row["ratio"] / row["truth"] - 1) for row in rows if row["check"] == kind]] if values}
injection = {}
for fault, kind in FAULTS.items():
entry = injection.setdefault(fault, {"check": kind, "sites_with_a_target": 0, "caught": 0, "missed_on": []})
for name in clean_sites:
if not layouts.get((name, fault), {}).get("fault_applied"):
continue
entry["sites_with_a_target"] += 1
if flagged(audits[name, fault], kind) > flagged(audits[name, "clean"], kind):
entry["caught"] += 1
else:
entry["missed_on"].append(name)
untouched = {}
for kind in GEOMETRY_CHECKS:
results = [result for name in clean_sites for result in audits[name, "clean"] if result["check"] == kind]
if results:
untouched[kind] = {"checks": len(results), "flagged": sum(result["advice"] != "keep" for result in results),
"flag_rate": sum(result["advice"] != "keep" for result in results) / len(results)}
summary = {"sites_collected": clean_sites, "sites_failed": [name for name in SITES if name not in clean_sites],
"detection": detection, "calibration_sites": calibration_sites, "test_sites": test_sites,
"tolerance_percentile": TOLERANCE_PERCENTILE, "tolerance": tolerances, "pixel_relative_error": relative_error,
"pixel_verdicts_on_test_sites": {"without_tolerance": agreement(test_rows, {}), "with_tolerance": agreement(test_rows, tolerances)},
"fault_injection": injection, "geometry_flags_on_untouched_sites": untouched}
(OUTPUT / "report.json").write_text(json.dumps(summary, indent=1), encoding="utf-8")
(PACKAGE / "tolerances.json").write_text(json.dumps({"source": "generalisation/report.json", "percentile": TOLERANCE_PERCENTILE,
"tolerance": tolerances}, indent=1), encoding="utf-8")
print("sites", len(clean_sites), "failed", summary["sites_failed"])
print("groups: same-class", sum(entry["same_class_card_groups"] for entry in detection.values()),
"class-free", sum(entry["repeated_groups"] for entry in detection.values()),
"| sites with any:", sum(entry["same_class_card_groups"] > 0 for entry in detection.values()),
"->", sum(entry["repeated_groups"] > 0 for entry in detection.values()))
print("tolerance", {kind: round(value, 3) for kind, value in tolerances.items()})
print("relative error", {kind: (entry["checks"], round(entry["median"], 3), round(entry["percentile_80"], 3)) for kind, entry in relative_error.items()})
for label, table in summary["pixel_verdicts_on_test_sites"].items():
print(label)
for kind, entry in table.items():
print(" ", kind, entry)
for fault, entry in injection.items():
print("fault", fault, entry)
for kind, entry in untouched.items():
print("untouched", kind, entry["checks"], entry["flagged"], round(entry["flag_rate"], 3))
if __name__ == "__main__":
collect() if sys.argv[1] == "collect" else report()
|