Download benchmark.py from codepawl/loupe: direct link, hf CLI and curl.
- Browser
- Download file 16.6 kB
-
https://huggingface.co/codepawl/loupe/resolve/main/benchmark.py
- Command line
-
hf download hf://codepawl/loupe/benchmark.py
-
curl -L -o benchmark.py https://huggingface.co/codepawl/loupe/resolve/main/benchmark.py
16.6 kB
| """Loupe's fixed benchmark, and the same faults shown to Claude Sonnet as screenshots. | |
| Every check has one fault that is injected into live pages (collect_page.cjs). A check is ready when it catches | |
| at least 90 percent of its injected faults and flags at most 3 percent of what it checks on untouched pages. | |
| The pages are the "third" set: collected after the checks were written, never used to tune anything. | |
| The versus part cuts a 1440 x 900 window around each fault, from the faulty page and from the untouched page, | |
| and asks Sonnet which of the ten issues it sees. Loupe answers the same question from its audit: an issue is | |
| "seen" when a check of that kind is flagged for something inside the window. Sonnet gets the picture only; | |
| Loupe gets the rendered page. That is the real difference between the two, and it is not corrected for. | |
| Usage: benchmark.py collect | report | crops | ask | versus | |
| """ | |
| import json | |
| import re | |
| import subprocess | |
| import sys | |
| import time | |
| from concurrent.futures import ThreadPoolExecutor | |
| from pathlib import Path | |
| sys.stdout.reconfigure(encoding="utf-8", errors="replace") | |
| PACKAGE = Path(__file__).resolve().parent | |
| sys.path.insert(0, str(PACKAGE)) | |
| from PIL import Image | |
| import evaluate_generalisation as evaluation | |
| Image.MAX_IMAGE_PIXELS = None | |
| OUTPUT = PACKAGE / "benchmark" | |
| # aljazeera never loads here (three sets of timeouts), so it is left out. | |
| PAGES = {name: url for name, url in evaluation.SETS["third"][0].items() if not name.startswith("aljazeera")} | |
| # fault: (check it must trip, the issue as shown to Sonnet) | |
| FAULTS = { | |
| "shift": ("row_alignment", "one item in a row of cards or tiles sits clearly lower or higher than its neighbours"), | |
| "gap": ("row_gap_evenness", "the horizontal gaps between the items of one row are clearly uneven, or two items overlap"), | |
| "height": ("item_height_evenness", "one item in a row is clearly shorter or taller than its neighbours"), | |
| "narrow": ("item_width_evenness", "one item in a row is clearly narrower or wider than its neighbours"), | |
| "squash": ("image_shape_evenness", "one picture in a row of cards has a different shape from its neighbours (squashed or stretched)"), | |
| "indent": ("text_alignment", "a heading is not lined up with the paragraph under it (neither left edges nor centres match)"), | |
| "small_heading": ("type_scale", "a heading is hardly larger than its paragraph (under about 1.3 times its size)"), | |
| "tight_lines": ("line_spacing", "a paragraph's lines are too tight (line height under about 1.2 times the font size)"), | |
| "low_contrast": ("text_contrast", "a paragraph's text has too little contrast against its background (under 4.5 to 1)"), | |
| "small_button": ("tap_target", "a button is too short to tap comfortably (under about 40 px tall)"), | |
| "hscroll": ("page_overflow", "something is wider than the page, so it runs off the right edge"), | |
| "clip": ("text_clipped", "a heading or text is cut off by its box"), | |
| "overlap": ("text_overlap", "two blocks of text overlap each other"), | |
| "far_heading": ("heading_proximity", "a heading sits closer to the block above it than to its own text below"), | |
| "distort": ("image_distortion", "a picture is squashed or stretched out of its natural shape"), | |
| "big_heading": ("heading_size_consistency", "headings of the same level have clearly different sizes"), | |
| "near_miss": ("edge_near_miss", "a text block's left edge is a few pixels off the edge its neighbours share"), | |
| "small_body": ("body_font_size", "a paragraph's text is smaller than about 16 px"), | |
| "light_heading": ("weight_contrast", "a heading is no bolder than the text under it"), | |
| "edge": ("edge_margin", "a paragraph touches the left edge of the page with no margin"), | |
| "small_title": ("title_is_largest", "the page's main heading is not the largest text"), | |
| "low_title": ("title_in_first_screen", "the page's main heading is pushed below the first screen"), | |
| } | |
| TEXT_FAULTS = ("indent", "small_heading", "tight_lines", "low_contrast", "small_button", "clip", "overlap", "far_heading", "near_miss", "small_body", | |
| "light_heading", "edge", "small_title", "low_title") | |
| # The thresholds named in the issue list above, so Loupe and Sonnet are asked the same question. | |
| BRIEF_CHANGES = {"type_scale": [1.3, 4.0], "line_spacing": [1.2, 2.2], "tap_target": [0.9, 3.0]} | |
| RECALL_GATE = 0.90 | |
| # An chose 10 percent on 2026-10-05: with 3 percent only three checks passed and the tool said nothing about a page | |
| # that was plainly broken. The measurements are exact (browser values) or validated (contrast), so most flags on | |
| # untouched pages are true statements about pages that do not follow the rule. | |
| CLEAN_GATE = 0.10 | |
| WINDOW = (1440, 900) | |
| PAGES_PER_FAULT = 8 | |
| ISSUE_IDS = list(FAULTS) | |
| def collect(): | |
| evaluation.SITES = PAGES | |
| evaluation.OUTPUT = OUTPUT | |
| evaluation.FAULTS = {fault: check for fault, (check, _) in FAULTS.items()} | |
| # Plain mode: one screenshot per page, no validation fields. | |
| original = evaluation.collect_one | |
| def plain(job): | |
| name, url, variant = job | |
| directory = OUTPUT / name / variant | |
| if (directory / "layout.json").exists(): | |
| return name, variant, "kept" | |
| arguments = ["node", str(PACKAGE / "collect_page.cjs"), url, str(directory), "1440", "plain"] + ([variant] if variant != "clean" else []) | |
| try: | |
| completed = subprocess.run(arguments, capture_output=True, text=True, encoding="utf-8", timeout=180) | |
| except subprocess.TimeoutExpired: | |
| return name, variant, "timeout" | |
| return name, variant, "ok" if completed.returncode == 0 else completed.stderr.strip().splitlines()[-1][:160] | |
| evaluation.collect_one = plain | |
| evaluation.collect(("clean", *FAULTS)) | |
| evaluation.collect_one = original | |
| def audits(): | |
| """Loupe's results for every collected page and variant, with the benchmark's brief.""" | |
| import page_audit | |
| read = page_audit.load_read() | |
| brief = {**page_audit.DEFAULT_BRIEF, **BRIEF_CHANGES} | |
| results = {} | |
| layouts = {} | |
| for name in PAGES: | |
| for variant in ("clean", *FAULTS): | |
| directory = OUTPUT / name / variant | |
| if (directory / "layout.json").exists(): | |
| layouts[name, variant] = json.loads((directory / "layout.json").read_text(encoding="utf-8")) | |
| results[name, variant] = page_audit.judge(page_audit.measure(directory, "dom"), brief, read) | |
| return results, layouts | |
| def window_around(box, page): | |
| """A WINDOW-sized part of the page centred on the box, kept inside the page.""" | |
| left = min(max(box[0] + box[2] / 2 - WINDOW[0] / 2, 0), max(page["width"] - WINDOW[0], 0)) | |
| top = min(max(box[1] + box[3] / 2 - WINDOW[1] / 2, 0), max(page["height"] - WINDOW[1], 0)) | |
| return [int(left), int(top), WINDOW[0], WINDOW[1]] | |
| def inside(region, window): | |
| centre_x = region[0] + region[2] / 2 | |
| centre_y = region[1] + region[3] / 2 | |
| return window[0] <= centre_x <= window[0] + window[2] and window[1] <= centre_y <= window[1] + window[3] | |
| def flagged_in(results, kind, window): | |
| return [result for result in results if result["check"] == kind and result["advice"] != "keep" and (result["region"] is None or inside(result["region"], window))] | |
| def report(): | |
| results, layouts = audits() | |
| pages = [name for name in PAGES if (name, "clean") in results] | |
| table = {} | |
| for fault, (kind, _) in FAULTS.items(): | |
| entry = {"check": kind, "pages_with_a_target": 0, "already_flagged_untouched": 0, "cancelled_by_page": 0, "caught": 0, "missed_on": []} | |
| for name in pages: | |
| layout = layouts.get((name, fault)) | |
| if not layout or not layout.get("fault_applied"): | |
| continue | |
| # The page's own CSS can cancel an injected style; a fault that changed nothing is not a fault. | |
| # A page that hides sideways overflow does not get wider either: the strip is cut off, not scrolled to. | |
| unchanged_width = fault == "hscroll" and layout["page"]["width"] <= layouts[name, "clean"]["page"]["width"] | |
| if layout.get("fault_changed_elements") == 0 or unchanged_width: | |
| entry["cancelled_by_page"] += 1 | |
| continue | |
| window = window_around(layout["fault_box"], layout["page"]) | |
| if fault in TEXT_FAULTS: | |
| on_target = lambda variant: any(result["target"] == layout["fault_text"] for result in flagged_in(results[name, variant], kind, window)) | |
| if on_target("clean"): | |
| entry["already_flagged_untouched"] += 1 | |
| continue | |
| caught = on_target(fault) | |
| else: | |
| caught = len(flagged_in(results[name, fault], kind, window)) > len(flagged_in(results[name, "clean"], kind, window)) | |
| entry["pages_with_a_target"] += 1 | |
| entry["caught"] += caught | |
| if not caught: | |
| entry["missed_on"].append(name) | |
| checked = [result for name in pages for result in results[name, "clean"] if result["check"] == kind] | |
| entry["untouched_checks"] = len(checked) | |
| entry["untouched_flagged"] = sum(result["advice"] != "keep" for result in checked) | |
| entry["recall"] = entry["caught"] / entry["pages_with_a_target"] if entry["pages_with_a_target"] else None | |
| entry["untouched_flag_rate"] = entry["untouched_flagged"] / len(checked) if checked else None | |
| entry["ready"] = bool(entry["recall"] is not None and entry["untouched_flag_rate"] is not None | |
| and entry["recall"] >= RECALL_GATE and entry["untouched_flag_rate"] <= CLEAN_GATE) | |
| table[fault] = entry | |
| summary = {"pages": pages, "pages_failed": [name for name in PAGES if name not in pages], "brief_changes": BRIEF_CHANGES, | |
| "gates": {"recall_at_least": RECALL_GATE, "untouched_flag_rate_at_most": CLEAN_GATE}, "faults": table} | |
| (OUTPUT / "report.json").write_text(json.dumps(summary, indent=1), encoding="utf-8") | |
| # The audit reads this list: a check is trusted only while it passes both gates here. | |
| (PACKAGE / "trusted_checks.json").write_text(json.dumps({"source": "benchmark/report.json", "gates": summary["gates"], | |
| "trusted": sorted(entry["check"] for entry in table.values() if entry["ready"])}, indent=1), encoding="utf-8") | |
| print("pages", len(pages), "failed", summary["pages_failed"]) | |
| for fault, entry in table.items(): | |
| print(f"{fault:14} {entry['check']:22} caught {entry['caught']}/{entry['pages_with_a_target']} " | |
| f"untouched {entry['untouched_flagged']}/{entry['untouched_checks']} " | |
| f"({(entry['untouched_flag_rate'] or 0) * 100:.1f}%) ready {entry['ready']} skipped {entry['already_flagged_untouched']} cancelled {entry['cancelled_by_page']} missed {entry['missed_on']}") | |
| def crops(): | |
| """A faulty and an untouched window for up to PAGES_PER_FAULT pages per fault, and the list of them.""" | |
| directory = OUTPUT / "versus" | |
| directory.mkdir(exist_ok=True) | |
| items = [] | |
| for fault in FAULTS: | |
| used = 0 | |
| for name in PAGES: | |
| layout_path = OUTPUT / name / fault / "layout.json" | |
| if used >= PAGES_PER_FAULT or not layout_path.exists() or not (OUTPUT / name / "clean" / "page.png").exists(): | |
| continue | |
| layout = json.loads(layout_path.read_text(encoding="utf-8")) | |
| if not layout.get("fault_applied"): | |
| continue | |
| window = window_around(layout["fault_box"], layout["page"]) | |
| for variant in (fault, "clean"): | |
| with Image.open(OUTPUT / name / variant / "page.png") as image: | |
| if window[1] + window[3] > image.height: | |
| continue | |
| image.convert("RGB").crop((window[0], window[1], window[0] + window[2], window[1] + window[3])).save(directory / f"{name}__{fault}__{variant}.png") | |
| items.append({"page": name, "fault": fault, "variant": variant, "window": window, "image": f"{name}__{fault}__{variant}.png"}) | |
| used += 1 | |
| (directory / "items.json").write_text(json.dumps(items, indent=1), encoding="utf-8") | |
| print(len(items), "windows;", {fault: sum(item["fault"] == fault and item["variant"] == "clean" for item in items) for fault in FAULTS}) | |
| def prompt_for(image): | |
| issues = "\n".join(f"{index + 1}. {FAULTS[fault][1]}" for index, fault in enumerate(ISSUE_IDS)) | |
| return (f"Open the image file {image} in the current folder with the Read tool. It is part of a web page, 1440 px wide, at actual size. " | |
| "Looking only at the picture, decide for each issue below whether it is present. Any number of them may be present, including none.\n\n" | |
| f"{issues}\n\nReply with one line of JSON and nothing else, listing the numbers that are present, like {{\"present\": [2, 7]}} or {{\"present\": []}}. " | |
| "Do not open any other file.") | |
| def ask_one(item): | |
| directory = OUTPUT / "versus" | |
| answer_path = directory / (item["image"][:-4] + ".answer.txt") | |
| if answer_path.exists(): | |
| return item["image"], "kept" | |
| started = time.perf_counter() | |
| completed = subprocess.run(["claude.cmd", "-p", "--model", "sonnet", "--allowedTools", "Read", "--strict-mcp-config"], input=prompt_for(item["image"]), | |
| capture_output=True, text=True, encoding="utf-8", cwd=str(directory), timeout=600) | |
| if completed.returncode != 0 or not completed.stdout.strip(): | |
| return item["image"], f"failed {completed.stderr[:120]}" | |
| answer_path.write_text(completed.stdout, encoding="utf-8") | |
| (directory / (item["image"][:-4] + ".seconds.txt")).write_text(f"{time.perf_counter() - started:.1f}", encoding="utf-8") | |
| return item["image"], "ok" | |
| def ask(): | |
| directory = OUTPUT / "versus" | |
| items = json.loads((directory / "items.json").read_text(encoding="utf-8")) | |
| (directory / "prompt_example.txt").write_text(prompt_for(items[0]["image"]), encoding="utf-8") | |
| with ThreadPoolExecutor(max_workers=3) as pool: | |
| for image, status in pool.map(ask_one, items): | |
| print(image, status, flush=True) | |
| def versus(): | |
| directory = OUTPUT / "versus" | |
| items = json.loads((directory / "items.json").read_text(encoding="utf-8")) | |
| results, _ = audits() | |
| table = {fault: {"windows": 0, "sonnet_hit": 0, "sonnet_untouched": 0, "loupe_hit": 0, "loupe_untouched": 0} for fault in FAULTS} | |
| seconds = [] | |
| unreadable = 0 | |
| for item in items: | |
| answer_path = directory / (item["image"][:-4] + ".answer.txt") | |
| if not answer_path.exists(): | |
| continue | |
| match = re.search(r"\{[^{}]*\"present\"\s*:\s*\[([^\]]*)\][^{}]*\}", answer_path.read_text(encoding="utf-8")) | |
| if match is None: | |
| unreadable += 1 | |
| continue | |
| named = {ISSUE_IDS[int(number) - 1] for number in re.findall(r"\d+", match.group(1)) if 1 <= int(number) <= len(ISSUE_IDS)} | |
| seconds.append(float((directory / (item["image"][:-4] + ".seconds.txt")).read_text(encoding="utf-8"))) | |
| fault = item["fault"] | |
| loupe = bool(flagged_in(results[item["page"], item["variant"]], FAULTS[fault][0], item["window"])) | |
| entry = table[fault] | |
| if item["variant"] == "clean": | |
| entry["sonnet_untouched"] += fault in named | |
| entry["loupe_untouched"] += loupe | |
| else: | |
| entry["windows"] += 1 | |
| entry["sonnet_hit"] += fault in named | |
| entry["loupe_hit"] += loupe | |
| summary = {"answers_unreadable": unreadable, "sonnet_seconds_median": sorted(seconds)[len(seconds) // 2] if seconds else None, "faults": table} | |
| (directory / "report.json").write_text(json.dumps(summary, indent=1), encoding="utf-8") | |
| print("answers", len(seconds), "unreadable", unreadable, "sonnet median seconds", summary["sonnet_seconds_median"]) | |
| print(f"{'fault':14} windows | Sonnet: says yes on faulty / on untouched | Loupe: same") | |
| for fault, entry in table.items(): | |
| print(f"{fault:14} {entry['windows']:3} | {entry['sonnet_hit']:2} / {entry['sonnet_untouched']:2} | {entry['loupe_hit']:2} / {entry['loupe_untouched']:2}") | |
| if __name__ == "__main__": | |
| {"collect": collect, "report": report, "crops": crops, "ask": ask, "versus": versus}[sys.argv[1]]() | |