File size: 16,027 Bytes
84c3565
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
"""How well the page audit holds on live public sites it was never tuned on.

Three questions, each answered with a number:
1. Detection: how many repeated groups the class-free detector finds per site, against the old same-class one.
2. Pixel checks against the browser: for every check read from pixels, the browser's own ratio gives the true
   verdict against the default brief. Reported as false flags and misses, without tolerance and with a tolerance
   calibrated on half of the sites and tested on the other half.
3. Geometry checks by fault injection: one known fault is applied to a live page (an item shifted down, pushed
   sideways, made shorter; a heading indented) and the audit must flag the matching check where the untouched
   page did not. The flag rate on untouched pages is reported next to it.

The "fresh" set was collected after every threshold, tolerance and scale had been fixed on the first set
(24 home pages). Nothing is calibrated on it: it only reports. Its contrast reference is not the declared
colours but the page rendered twice, with and without text.

Usage: evaluate_generalisation.py collect|report [first|fresh]
"""

import json

import numpy
import subprocess
import sys
from concurrent.futures import ThreadPoolExecutor
from pathlib import Path

sys.stdout.reconfigure(encoding="utf-8", errors="replace")
PACKAGE = Path(__file__).resolve().parent
sys.path.insert(0, str(PACKAGE))
OUTPUT = PACKAGE / "generalisation"
SITES = {
    "python": "https://www.python.org/", "rust": "https://www.rust-lang.org/", "django": "https://www.djangoproject.com/",
    "bootstrap": "https://getbootstrap.com/", "tailwind": "https://tailwindcss.com/", "vue": "https://vuejs.org/",
    "svelte": "https://svelte.dev/", "node": "https://nodejs.org/en", "mozilla": "https://www.mozilla.org/en-US/",
    "govuk": "https://www.gov.uk/", "wikipedia": "https://en.wikipedia.org/wiki/Typography", "mdn": "https://developer.mozilla.org/en-US/",
    "github": "https://github.com/features", "stripe": "https://stripe.com/", "shopify": "https://www.shopify.com/",
    "ikea": "https://www.ikea.com/us/en/", "bbc": "https://www.bbc.com/", "guardian": "https://www.theguardian.com/international",
    "nasa": "https://www.nasa.gov/", "mit": "https://www.mit.edu/", "figma": "https://www.figma.com/",
    "notion": "https://www.notion.com/", "vercel": "https://vercel.com/", "astro": "https://astro.build/",
}
FRESH = {
    "apple": ["https://www.apple.com/", "https://www.apple.com/mac/", "https://www.apple.com/iphone/"],
    "ruby": ["https://www.ruby-lang.org/en/", "https://www.ruby-lang.org/en/documentation/", "https://www.ruby-lang.org/en/downloads/"],
    "go": ["https://go.dev/", "https://go.dev/learn/", "https://go.dev/solutions/"],
    "postgres": ["https://www.postgresql.org/", "https://www.postgresql.org/about/", "https://www.postgresql.org/download/"],
    "kotlin": ["https://kotlinlang.org/", "https://kotlinlang.org/docs/home.html", "https://kotlinlang.org/community/"],
    "typescript": ["https://www.typescriptlang.org/", "https://www.typescriptlang.org/docs/", "https://www.typescriptlang.org/download/"],
    "w3c": ["https://www.w3.org/", "https://www.w3.org/standards/", "https://www.w3.org/about/"],
    "npr": ["https://www.npr.org/", "https://www.npr.org/sections/news/", "https://www.npr.org/sections/culture/"],
    "allbirds": ["https://www.allbirds.com/", "https://www.allbirds.com/collections/mens", "https://www.allbirds.com/pages/our-story"],
    "harvard": ["https://www.harvard.edu/", "https://www.harvard.edu/about/", "https://www.harvard.edu/academics/"],
    "docker": ["https://www.docker.com/", "https://www.docker.com/products/docker-desktop/", "https://www.docker.com/pricing/"],
    "slack": ["https://slack.com/", "https://slack.com/features", "https://slack.com/pricing"],
}
THIRD = {
    "php": ["https://www.php.net/", "https://www.php.net/docs.php", "https://www.php.net/downloads"],
    "elixir": ["https://elixir-lang.org/", "https://elixir-lang.org/learning.html", "https://elixir-lang.org/install.html"],
    "raspberrypi": ["https://www.raspberrypi.com/", "https://www.raspberrypi.com/products/", "https://www.raspberrypi.com/software/"],
    "mongodb": ["https://www.mongodb.com/", "https://www.mongodb.com/pricing", "https://www.mongodb.com/company"],
    "dropbox": ["https://www.dropbox.com/", "https://www.dropbox.com/features", "https://www.dropbox.com/plans"],
    "stanford": ["https://www.stanford.edu/", "https://www.stanford.edu/about/", "https://www.stanford.edu/academics/"],
    "who": ["https://www.who.int/", "https://www.who.int/about", "https://www.who.int/health-topics"],
    "aljazeera": ["https://www.aljazeera.com/", "https://www.aljazeera.com/news/", "https://www.aljazeera.com/sports/"],
    "gitlab": ["https://about.gitlab.com/", "https://about.gitlab.com/pricing/", "https://about.gitlab.com/platform/"],
    "cloudflare": ["https://www.cloudflare.com/", "https://www.cloudflare.com/plans/", "https://www.cloudflare.com/learning/"],
    "zendesk": ["https://www.zendesk.com/", "https://www.zendesk.com/pricing/", "https://www.zendesk.com/service/"],
    "muji": ["https://www.muji.us/", "https://www.muji.us/collections/stationery", "https://www.muji.us/pages/about-muji"],
}
pages_of = lambda sites: {f"{site}-{index + 1}": url for site, urls in sites.items() for index, url in enumerate(urls)}
SETS = {"first": (SITES, PACKAGE / "generalisation"),
        # The fresh pages collected again once line boxes existed: the set the line-box reading was developed on.
        "dev": (pages_of(FRESH), PACKAGE / "generalisation_dev"),
        # Pages first collected after the line-box reading was fixed: its unseen test.
        "third": (pages_of(THIRD), PACKAGE / "generalisation_third"),
        "fresh": ({f"{site}-{index + 1}": url for site, urls in FRESH.items() for index, url in enumerate(urls)}, PACKAGE / "generalisation_fresh")}
FAULTS = {"shift": "row_alignment", "gap": "row_gap_evenness", "height": "item_height_evenness", "indent": "text_alignment"}
PIXEL_CHECKS = ("type_scale", "line_spacing", "line_length", "text_contrast")
GEOMETRY_CHECKS = ("text_alignment", "row_alignment", "row_gap_evenness", "item_height_evenness", "item_width_evenness",
                   "image_shape_evenness", "column_alignment", "column_gap_evenness")
TOLERANCE_PERCENTILE = 80


def collect_one(job):
    name, url, variant = job
    directory = OUTPUT / name / variant
    if (directory / "layout.json").exists():
        return name, variant, "kept"
    arguments = ["node", str(PACKAGE / "collect_page.cjs"), url, str(directory), "1440", "truth"]
    if variant != "clean":
        arguments.append(variant)
    try:
        completed = subprocess.run(arguments, capture_output=True, text=True, encoding="utf-8", timeout=180)
    except subprocess.TimeoutExpired:
        return name, variant, "timeout"
    return name, variant, "ok" if completed.returncode == 0 else completed.stderr.strip().splitlines()[-1][:160]


def collect(variants=("clean", *FAULTS)):
    jobs = [(name, url, variant) for name, url in SITES.items() for variant in variants]
    with ThreadPoolExecutor(max_workers=4) as pool:
        for name, variant, status in pool.map(collect_one, jobs):
            print(name, variant, status, flush=True)


def rendered_contrast(pixels, without_text, box):
    """Contrast of a text block from the page rendered with and without its text; None when the two differ too little or too much."""
    import block_measure

    left, top, width, height = (int(round(value)) for value in box)
    painted = pixels[max(top, 0):top + height, max(left, 0):left + width].astype(numpy.int32)
    behind = without_text[max(top, 0):top + height, max(left, 0):left + width].astype(numpy.int32)
    if painted.shape != behind.shape or painted.size == 0:
        return None
    difference = numpy.abs(painted - behind).sum(axis=2)
    glyphs = difference > 24
    # Under a dozen changed pixels is no text; most of the box changing means the page moved between the two pictures.
    if glyphs.sum() < 12 or glyphs.mean() > 0.7:
        return None
    core = difference >= numpy.percentile(difference[glyphs], 90)
    return block_measure.contrast_ratio(behind[core].mean(axis=0), painted[core].mean(axis=0))


def true_ratio(result, layout, pictures=None):
    """The same ratio from the browser's computed values (or, for contrast, from the two renderings), or None."""
    import validate_page_measures
    import block_measure

    truths = [layout["texts"][text_id]["truth"] for text_id in result["text_ids"]]
    if result["check"] == "type_scale":
        return truths[0]["font_px"] / truths[1]["font_px"]
    if result["check"] == "line_spacing":
        return truths[0]["line_height_px"] / truths[0]["font_px"] if truths[0]["line_height_px"] else None
    if result["check"] == "line_length":
        return result["numerator"]["value"] / truths[0]["font_px"]
    if pictures is not None:
        return rendered_contrast(pictures[0], pictures[1], layout["texts"][result["text_ids"][0]]["box"])
    behind = validate_page_measures.colour(truths[0]["background"])
    return block_measure.contrast_ratio(behind, validate_page_measures.colour(truths[0]["color"], behind))


def verdict(ratio, band, tolerance=0.0):
    if ratio < band[0] / (1 + tolerance):
        return "too low"
    return "too high" if ratio > band[1] * (1 + tolerance) else "in range"


def pixel_rows(name, results, layout, pictures=None):
    rows = []
    for result in results:
        if result["check"] not in PIXEL_CHECKS:
            continue
        truth = true_ratio(result, layout, pictures if result["check"] == "text_contrast" else None)
        if truth:
            rows.append({"site": name, "check": result["check"], "ratio": result["ratio"], "truth": truth, "band": result["band"]})
    return rows


def agreement(rows, tolerances):
    table = {}
    for row in rows:
        entry = table.setdefault(row["check"], {"checks": 0, "truly_out": 0, "false_flags": 0, "misses": 0, "wrong_side": 0})
        read = verdict(row["ratio"], row["band"], tolerances.get(row["check"], 0.0))
        real = verdict(row["truth"], row["band"])
        entry["checks"] += 1
        entry["truly_out"] += real != "in range"
        entry["false_flags"] += read != "in range" and real == "in range"
        entry["misses"] += read == "in range" and real != "in range"
        entry["wrong_side"] += "in range" not in (read, real) and read != real
    for entry in table.values():
        truly_in = entry["checks"] - entry["truly_out"]
        entry["false_flag_rate"] = entry["false_flags"] / truly_in if truly_in else None
        entry["miss_rate"] = entry["misses"] / entry["truly_out"] if entry["truly_out"] else None
    return table


def flagged(results, kind):
    return sum(result["check"] == kind and result["advice"] != "keep" for result in results)


def report():
    import numpy
    import page_audit

    read = page_audit.load_read()
    audits = {}
    layouts = {}
    for name in SITES:
        for variant in ("clean", *FAULTS):
            directory = OUTPUT / name / variant
            if not (directory / "layout.json").exists():
                continue
            layouts[name, variant] = json.loads((directory / "layout.json").read_text(encoding="utf-8"))
            audits[name, variant] = page_audit.judge(page_audit.measure(directory), page_audit.DEFAULT_BRIEF, read)
    clean_sites = [name for name in SITES if (name, "clean") in audits]

    detection = {name: {"texts": len(layouts[name, "clean"]["texts"]),
                        "same_class_card_groups": len(layouts[name, "clean"]["card_groups"]),
                        "repeated_groups": len(layouts[name, "clean"]["repeats"]),
                        "checks": len(audits[name, "clean"])} for name in clean_sites}

    calibration_sites, test_sites = clean_sites[0::2], clean_sites[1::2]
    rows = [row for name in clean_sites for row in pixel_rows(name, audits[name, "clean"], layouts[name, "clean"])]
    errors = {}
    for row in rows:
        if row["site"] in calibration_sites:
            errors.setdefault(row["check"], []).append(abs(row["ratio"] / row["truth"] - 1))
    tolerances = {kind: float(numpy.percentile(values, TOLERANCE_PERCENTILE)) for kind, values in errors.items()}
    # Contrast is read exactly where the text sits on a flat colour and is far off elsewhere (text over pictures),
    # so a tolerance cannot describe it; it keeps none.
    tolerances["text_contrast"] = 0.0
    test_rows = [row for row in rows if row["site"] in test_sites]
    relative_error = {kind: {"checks": len(values), "median": float(numpy.median(values)), "percentile_80": float(numpy.percentile(values, 80))}
                      for kind in PIXEL_CHECKS
                      for values in [[abs(row["ratio"] / row["truth"] - 1) for row in rows if row["check"] == kind]] if values}

    injection = {}
    for fault, kind in FAULTS.items():
        entry = injection.setdefault(fault, {"check": kind, "sites_with_a_target": 0, "caught": 0, "missed_on": []})
        for name in clean_sites:
            if not layouts.get((name, fault), {}).get("fault_applied"):
                continue
            entry["sites_with_a_target"] += 1
            if flagged(audits[name, fault], kind) > flagged(audits[name, "clean"], kind):
                entry["caught"] += 1
            else:
                entry["missed_on"].append(name)
    untouched = {}
    for kind in GEOMETRY_CHECKS:
        results = [result for name in clean_sites for result in audits[name, "clean"] if result["check"] == kind]
        if results:
            untouched[kind] = {"checks": len(results), "flagged": sum(result["advice"] != "keep" for result in results),
                               "flag_rate": sum(result["advice"] != "keep" for result in results) / len(results)}

    summary = {"sites_collected": clean_sites, "sites_failed": [name for name in SITES if name not in clean_sites],
               "detection": detection, "calibration_sites": calibration_sites, "test_sites": test_sites,
               "tolerance_percentile": TOLERANCE_PERCENTILE, "tolerance": tolerances, "pixel_relative_error": relative_error,
               "pixel_verdicts_on_test_sites": {"without_tolerance": agreement(test_rows, {}), "with_tolerance": agreement(test_rows, tolerances)},
               "fault_injection": injection, "geometry_flags_on_untouched_sites": untouched}
    (OUTPUT / "report.json").write_text(json.dumps(summary, indent=1), encoding="utf-8")
    (PACKAGE / "tolerances.json").write_text(json.dumps({"source": "generalisation/report.json", "percentile": TOLERANCE_PERCENTILE,
                                                         "tolerance": tolerances}, indent=1), encoding="utf-8")
    print("sites", len(clean_sites), "failed", summary["sites_failed"])
    print("groups: same-class", sum(entry["same_class_card_groups"] for entry in detection.values()),
          "class-free", sum(entry["repeated_groups"] for entry in detection.values()),
          "| sites with any:", sum(entry["same_class_card_groups"] > 0 for entry in detection.values()),
          "->", sum(entry["repeated_groups"] > 0 for entry in detection.values()))
    print("tolerance", {kind: round(value, 3) for kind, value in tolerances.items()})
    print("relative error", {kind: (entry["checks"], round(entry["median"], 3), round(entry["percentile_80"], 3)) for kind, entry in relative_error.items()})
    for label, table in summary["pixel_verdicts_on_test_sites"].items():
        print(label)
        for kind, entry in table.items():
            print("  ", kind, entry)
    for fault, entry in injection.items():
        print("fault", fault, entry)
    for kind, entry in untouched.items():
        print("untouched", kind, entry["checks"], entry["flagged"], round(entry["flag_rate"], 3))


if __name__ == "__main__":
    collect() if sys.argv[1] == "collect" else report()