loupe / benchmark.py
anxuanng's picture
Gate chart: one-line note
5fd9805 verified
Raw History Blame Contribute Delete
16.6 kB
"""Loupe's fixed benchmark, and the same faults shown to Claude Sonnet as screenshots.
Every check has one fault that is injected into live pages (collect_page.cjs). A check is ready when it catches
at least 90 percent of its injected faults and flags at most 3 percent of what it checks on untouched pages.
The pages are the "third" set: collected after the checks were written, never used to tune anything.
The versus part cuts a 1440 x 900 window around each fault, from the faulty page and from the untouched page,
and asks Sonnet which of the ten issues it sees. Loupe answers the same question from its audit: an issue is
"seen" when a check of that kind is flagged for something inside the window. Sonnet gets the picture only;
Loupe gets the rendered page. That is the real difference between the two, and it is not corrected for.
Usage: benchmark.py collect | report | crops | ask | versus
"""
import json
import re
import subprocess
import sys
import time
from concurrent.futures import ThreadPoolExecutor
from pathlib import Path
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
PACKAGE = Path(__file__).resolve().parent
sys.path.insert(0, str(PACKAGE))
from PIL import Image
import evaluate_generalisation as evaluation
Image.MAX_IMAGE_PIXELS = None
OUTPUT = PACKAGE / "benchmark"
# aljazeera never loads here (three sets of timeouts), so it is left out.
PAGES = {name: url for name, url in evaluation.SETS["third"][0].items() if not name.startswith("aljazeera")}
# fault: (check it must trip, the issue as shown to Sonnet)
FAULTS = {
"shift": ("row_alignment", "one item in a row of cards or tiles sits clearly lower or higher than its neighbours"),
"gap": ("row_gap_evenness", "the horizontal gaps between the items of one row are clearly uneven, or two items overlap"),
"height": ("item_height_evenness", "one item in a row is clearly shorter or taller than its neighbours"),
"narrow": ("item_width_evenness", "one item in a row is clearly narrower or wider than its neighbours"),
"squash": ("image_shape_evenness", "one picture in a row of cards has a different shape from its neighbours (squashed or stretched)"),
"indent": ("text_alignment", "a heading is not lined up with the paragraph under it (neither left edges nor centres match)"),
"small_heading": ("type_scale", "a heading is hardly larger than its paragraph (under about 1.3 times its size)"),
"tight_lines": ("line_spacing", "a paragraph's lines are too tight (line height under about 1.2 times the font size)"),
"low_contrast": ("text_contrast", "a paragraph's text has too little contrast against its background (under 4.5 to 1)"),
"small_button": ("tap_target", "a button is too short to tap comfortably (under about 40 px tall)"),
"hscroll": ("page_overflow", "something is wider than the page, so it runs off the right edge"),
"clip": ("text_clipped", "a heading or text is cut off by its box"),
"overlap": ("text_overlap", "two blocks of text overlap each other"),
"far_heading": ("heading_proximity", "a heading sits closer to the block above it than to its own text below"),
"distort": ("image_distortion", "a picture is squashed or stretched out of its natural shape"),
"big_heading": ("heading_size_consistency", "headings of the same level have clearly different sizes"),
"near_miss": ("edge_near_miss", "a text block's left edge is a few pixels off the edge its neighbours share"),
"small_body": ("body_font_size", "a paragraph's text is smaller than about 16 px"),
"light_heading": ("weight_contrast", "a heading is no bolder than the text under it"),
"edge": ("edge_margin", "a paragraph touches the left edge of the page with no margin"),
"small_title": ("title_is_largest", "the page's main heading is not the largest text"),
"low_title": ("title_in_first_screen", "the page's main heading is pushed below the first screen"),
}
TEXT_FAULTS = ("indent", "small_heading", "tight_lines", "low_contrast", "small_button", "clip", "overlap", "far_heading", "near_miss", "small_body",
"light_heading", "edge", "small_title", "low_title")
# The thresholds named in the issue list above, so Loupe and Sonnet are asked the same question.
BRIEF_CHANGES = {"type_scale": [1.3, 4.0], "line_spacing": [1.2, 2.2], "tap_target": [0.9, 3.0]}
RECALL_GATE = 0.90
# An chose 10 percent on 2026-10-05: with 3 percent only three checks passed and the tool said nothing about a page
# that was plainly broken. The measurements are exact (browser values) or validated (contrast), so most flags on
# untouched pages are true statements about pages that do not follow the rule.
CLEAN_GATE = 0.10
WINDOW = (1440, 900)
PAGES_PER_FAULT = 8
ISSUE_IDS = list(FAULTS)
def collect():
evaluation.SITES = PAGES
evaluation.OUTPUT = OUTPUT
evaluation.FAULTS = {fault: check for fault, (check, _) in FAULTS.items()}
# Plain mode: one screenshot per page, no validation fields.
original = evaluation.collect_one
def plain(job):
name, url, variant = job
directory = OUTPUT / name / variant
if (directory / "layout.json").exists():
return name, variant, "kept"
arguments = ["node", str(PACKAGE / "collect_page.cjs"), url, str(directory), "1440", "plain"] + ([variant] if variant != "clean" else [])
try:
completed = subprocess.run(arguments, capture_output=True, text=True, encoding="utf-8", timeout=180)
except subprocess.TimeoutExpired:
return name, variant, "timeout"
return name, variant, "ok" if completed.returncode == 0 else completed.stderr.strip().splitlines()[-1][:160]
evaluation.collect_one = plain
evaluation.collect(("clean", *FAULTS))
evaluation.collect_one = original
def audits():
"""Loupe's results for every collected page and variant, with the benchmark's brief."""
import page_audit
read = page_audit.load_read()
brief = {**page_audit.DEFAULT_BRIEF, **BRIEF_CHANGES}
results = {}
layouts = {}
for name in PAGES:
for variant in ("clean", *FAULTS):
directory = OUTPUT / name / variant
if (directory / "layout.json").exists():
layouts[name, variant] = json.loads((directory / "layout.json").read_text(encoding="utf-8"))
results[name, variant] = page_audit.judge(page_audit.measure(directory, "dom"), brief, read)
return results, layouts
def window_around(box, page):
"""A WINDOW-sized part of the page centred on the box, kept inside the page."""
left = min(max(box[0] + box[2] / 2 - WINDOW[0] / 2, 0), max(page["width"] - WINDOW[0], 0))
top = min(max(box[1] + box[3] / 2 - WINDOW[1] / 2, 0), max(page["height"] - WINDOW[1], 0))
return [int(left), int(top), WINDOW[0], WINDOW[1]]
def inside(region, window):
centre_x = region[0] + region[2] / 2
centre_y = region[1] + region[3] / 2
return window[0] <= centre_x <= window[0] + window[2] and window[1] <= centre_y <= window[1] + window[3]
def flagged_in(results, kind, window):
return [result for result in results if result["check"] == kind and result["advice"] != "keep" and (result["region"] is None or inside(result["region"], window))]
def report():
results, layouts = audits()
pages = [name for name in PAGES if (name, "clean") in results]
table = {}
for fault, (kind, _) in FAULTS.items():
entry = {"check": kind, "pages_with_a_target": 0, "already_flagged_untouched": 0, "cancelled_by_page": 0, "caught": 0, "missed_on": []}
for name in pages:
layout = layouts.get((name, fault))
if not layout or not layout.get("fault_applied"):
continue
# The page's own CSS can cancel an injected style; a fault that changed nothing is not a fault.
# A page that hides sideways overflow does not get wider either: the strip is cut off, not scrolled to.
unchanged_width = fault == "hscroll" and layout["page"]["width"] <= layouts[name, "clean"]["page"]["width"]
if layout.get("fault_changed_elements") == 0 or unchanged_width:
entry["cancelled_by_page"] += 1
continue
window = window_around(layout["fault_box"], layout["page"])
if fault in TEXT_FAULTS:
on_target = lambda variant: any(result["target"] == layout["fault_text"] for result in flagged_in(results[name, variant], kind, window))
if on_target("clean"):
entry["already_flagged_untouched"] += 1
continue
caught = on_target(fault)
else:
caught = len(flagged_in(results[name, fault], kind, window)) > len(flagged_in(results[name, "clean"], kind, window))
entry["pages_with_a_target"] += 1
entry["caught"] += caught
if not caught:
entry["missed_on"].append(name)
checked = [result for name in pages for result in results[name, "clean"] if result["check"] == kind]
entry["untouched_checks"] = len(checked)
entry["untouched_flagged"] = sum(result["advice"] != "keep" for result in checked)
entry["recall"] = entry["caught"] / entry["pages_with_a_target"] if entry["pages_with_a_target"] else None
entry["untouched_flag_rate"] = entry["untouched_flagged"] / len(checked) if checked else None
entry["ready"] = bool(entry["recall"] is not None and entry["untouched_flag_rate"] is not None
and entry["recall"] >= RECALL_GATE and entry["untouched_flag_rate"] <= CLEAN_GATE)
table[fault] = entry
summary = {"pages": pages, "pages_failed": [name for name in PAGES if name not in pages], "brief_changes": BRIEF_CHANGES,
"gates": {"recall_at_least": RECALL_GATE, "untouched_flag_rate_at_most": CLEAN_GATE}, "faults": table}
(OUTPUT / "report.json").write_text(json.dumps(summary, indent=1), encoding="utf-8")
# The audit reads this list: a check is trusted only while it passes both gates here.
(PACKAGE / "trusted_checks.json").write_text(json.dumps({"source": "benchmark/report.json", "gates": summary["gates"],
"trusted": sorted(entry["check"] for entry in table.values() if entry["ready"])}, indent=1), encoding="utf-8")
print("pages", len(pages), "failed", summary["pages_failed"])
for fault, entry in table.items():
print(f"{fault:14} {entry['check']:22} caught {entry['caught']}/{entry['pages_with_a_target']} "
f"untouched {entry['untouched_flagged']}/{entry['untouched_checks']} "
f"({(entry['untouched_flag_rate'] or 0) * 100:.1f}%) ready {entry['ready']} skipped {entry['already_flagged_untouched']} cancelled {entry['cancelled_by_page']} missed {entry['missed_on']}")
def crops():
"""A faulty and an untouched window for up to PAGES_PER_FAULT pages per fault, and the list of them."""
directory = OUTPUT / "versus"
directory.mkdir(exist_ok=True)
items = []
for fault in FAULTS:
used = 0
for name in PAGES:
layout_path = OUTPUT / name / fault / "layout.json"
if used >= PAGES_PER_FAULT or not layout_path.exists() or not (OUTPUT / name / "clean" / "page.png").exists():
continue
layout = json.loads(layout_path.read_text(encoding="utf-8"))
if not layout.get("fault_applied"):
continue
window = window_around(layout["fault_box"], layout["page"])
for variant in (fault, "clean"):
with Image.open(OUTPUT / name / variant / "page.png") as image:
if window[1] + window[3] > image.height:
continue
image.convert("RGB").crop((window[0], window[1], window[0] + window[2], window[1] + window[3])).save(directory / f"{name}__{fault}__{variant}.png")
items.append({"page": name, "fault": fault, "variant": variant, "window": window, "image": f"{name}__{fault}__{variant}.png"})
used += 1
(directory / "items.json").write_text(json.dumps(items, indent=1), encoding="utf-8")
print(len(items), "windows;", {fault: sum(item["fault"] == fault and item["variant"] == "clean" for item in items) for fault in FAULTS})
def prompt_for(image):
issues = "\n".join(f"{index + 1}. {FAULTS[fault][1]}" for index, fault in enumerate(ISSUE_IDS))
return (f"Open the image file {image} in the current folder with the Read tool. It is part of a web page, 1440 px wide, at actual size. "
"Looking only at the picture, decide for each issue below whether it is present. Any number of them may be present, including none.\n\n"
f"{issues}\n\nReply with one line of JSON and nothing else, listing the numbers that are present, like {{\"present\": [2, 7]}} or {{\"present\": []}}. "
"Do not open any other file.")
def ask_one(item):
directory = OUTPUT / "versus"
answer_path = directory / (item["image"][:-4] + ".answer.txt")
if answer_path.exists():
return item["image"], "kept"
started = time.perf_counter()
completed = subprocess.run(["claude.cmd", "-p", "--model", "sonnet", "--allowedTools", "Read", "--strict-mcp-config"], input=prompt_for(item["image"]),
capture_output=True, text=True, encoding="utf-8", cwd=str(directory), timeout=600)
if completed.returncode != 0 or not completed.stdout.strip():
return item["image"], f"failed {completed.stderr[:120]}"
answer_path.write_text(completed.stdout, encoding="utf-8")
(directory / (item["image"][:-4] + ".seconds.txt")).write_text(f"{time.perf_counter() - started:.1f}", encoding="utf-8")
return item["image"], "ok"
def ask():
directory = OUTPUT / "versus"
items = json.loads((directory / "items.json").read_text(encoding="utf-8"))
(directory / "prompt_example.txt").write_text(prompt_for(items[0]["image"]), encoding="utf-8")
with ThreadPoolExecutor(max_workers=3) as pool:
for image, status in pool.map(ask_one, items):
print(image, status, flush=True)
def versus():
directory = OUTPUT / "versus"
items = json.loads((directory / "items.json").read_text(encoding="utf-8"))
results, _ = audits()
table = {fault: {"windows": 0, "sonnet_hit": 0, "sonnet_untouched": 0, "loupe_hit": 0, "loupe_untouched": 0} for fault in FAULTS}
seconds = []
unreadable = 0
for item in items:
answer_path = directory / (item["image"][:-4] + ".answer.txt")
if not answer_path.exists():
continue
match = re.search(r"\{[^{}]*\"present\"\s*:\s*\[([^\]]*)\][^{}]*\}", answer_path.read_text(encoding="utf-8"))
if match is None:
unreadable += 1
continue
named = {ISSUE_IDS[int(number) - 1] for number in re.findall(r"\d+", match.group(1)) if 1 <= int(number) <= len(ISSUE_IDS)}
seconds.append(float((directory / (item["image"][:-4] + ".seconds.txt")).read_text(encoding="utf-8")))
fault = item["fault"]
loupe = bool(flagged_in(results[item["page"], item["variant"]], FAULTS[fault][0], item["window"]))
entry = table[fault]
if item["variant"] == "clean":
entry["sonnet_untouched"] += fault in named
entry["loupe_untouched"] += loupe
else:
entry["windows"] += 1
entry["sonnet_hit"] += fault in named
entry["loupe_hit"] += loupe
summary = {"answers_unreadable": unreadable, "sonnet_seconds_median": sorted(seconds)[len(seconds) // 2] if seconds else None, "faults": table}
(directory / "report.json").write_text(json.dumps(summary, indent=1), encoding="utf-8")
print("answers", len(seconds), "unreadable", unreadable, "sonnet median seconds", summary["sonnet_seconds_median"])
print(f"{'fault':14} windows | Sonnet: says yes on faulty / on untouched | Loupe: same")
for fault, entry in table.items():
print(f"{fault:14} {entry['windows']:3} | {entry['sonnet_hit']:2} / {entry['sonnet_untouched']:2} | {entry['loupe_hit']:2} / {entry['loupe_untouched']:2}")
if __name__ == "__main__":
{"collect": collect, "report": report, "crops": crops, "ask": ask, "versus": versus}[sys.argv[1]]()