Download sensitivity.py from codepawl/loupe: direct link, hf CLI and curl.
- Browser
- Download file 4.05 kB
-
https://huggingface.co/codepawl/loupe/resolve/main/sensitivity.py
- Command line
-
hf download hf://codepawl/loupe/sensitivity.py
-
curl -L -o sensitivity.py https://huggingface.co/codepawl/loupe/resolve/main/sensitivity.py
4.05 kB
| """How small a fault Loupe catches: three geometric faults at 2, 4, 8 and 16 px on the benchmark pages. | |
| The benchmark injects each fault at one size (22, 26 and 34 px). Here the same faults are injected smaller. | |
| A fault counts as caught by the same rule as benchmark.py: more flagged checks of its kind inside the window | |
| around it than on the untouched page (for the heading indent: the heading itself flagged, and not before). | |
| Usage: sensitivity.py collect | report | |
| """ | |
| import json | |
| import subprocess | |
| import sys | |
| from concurrent.futures import ThreadPoolExecutor | |
| from pathlib import Path | |
| sys.stdout.reconfigure(encoding="utf-8", errors="replace") | |
| PACKAGE = Path(__file__).resolve().parent | |
| sys.path.insert(0, str(PACKAGE)) | |
| import benchmark | |
| SIZES = (2, 4, 8, 16) | |
| FAULTS = ("shift", "gap", "indent") | |
| USUAL_SIZE = {"shift": 22, "gap": 26, "indent": 34} | |
| def collect_one(job): | |
| name, url, variant = job | |
| directory = benchmark.OUTPUT / name / variant.replace("@", "_at_") | |
| if (directory / "layout.json").exists(): | |
| return name, variant, "kept" | |
| try: | |
| completed = subprocess.run(["node", str(PACKAGE / "collect_page.cjs"), url, str(directory), "1440", "plain", variant], | |
| capture_output=True, text=True, encoding="utf-8", timeout=180) | |
| except subprocess.TimeoutExpired: | |
| return name, variant, "timeout" | |
| return name, variant, "ok" if completed.returncode == 0 else completed.stderr.strip().splitlines()[-1][:160] | |
| def collect(): | |
| jobs = [(name, url, f"{fault}@{size}") for name, url in benchmark.PAGES.items() for fault in FAULTS for size in SIZES] | |
| with ThreadPoolExecutor(max_workers=4) as pool: | |
| for name, variant, status in pool.map(collect_one, jobs): | |
| print(name, variant, status, flush=True) | |
| def report(): | |
| import page_audit | |
| read = page_audit.load_read() | |
| brief = {**page_audit.DEFAULT_BRIEF, **benchmark.BRIEF_CHANGES} | |
| judged = {} | |
| def results_of(name, variant): | |
| directory = benchmark.OUTPUT / name / variant | |
| if (name, variant) not in judged and (directory / "layout.json").exists(): | |
| judged[name, variant] = (page_audit.judge(page_audit.measure(directory, "dom"), brief, read), | |
| json.loads((directory / "layout.json").read_text(encoding="utf-8"))) | |
| return judged.get((name, variant)) | |
| table = {fault: {} for fault in FAULTS} | |
| for fault in FAULTS: | |
| kind = benchmark.FAULTS[fault][0] | |
| for size in (*SIZES, USUAL_SIZE[fault]): | |
| variant = fault if size == USUAL_SIZE[fault] else f"{fault}_at_{size}" | |
| entry = {"pages": 0, "caught": 0} | |
| for name in benchmark.PAGES: | |
| clean = results_of(name, "clean") | |
| faulty = results_of(name, variant) | |
| if not clean or not faulty or not faulty[1].get("fault_applied") or faulty[1].get("fault_changed_elements") == 0: | |
| continue | |
| window = benchmark.window_around(faulty[1]["fault_box"], faulty[1]["page"]) | |
| if fault == "indent": | |
| on_target = lambda results: any(result["target"] == faulty[1]["fault_text"] for result in benchmark.flagged_in(results, kind, window)) | |
| if on_target(clean[0]): | |
| continue | |
| caught = on_target(faulty[0]) | |
| else: | |
| caught = len(benchmark.flagged_in(faulty[0], kind, window)) > len(benchmark.flagged_in(clean[0], kind, window)) | |
| entry["pages"] += 1 | |
| entry["caught"] += caught | |
| table[fault][size] = entry | |
| (benchmark.OUTPUT / "sensitivity.json").write_text(json.dumps({"sizes_px": [*SIZES], "usual_size_px": USUAL_SIZE, "faults": table}, indent=1), encoding="utf-8") | |
| for fault, sizes in table.items(): | |
| print(fault, {size: f"{entry['caught']}/{entry['pages']}" for size, entry in sizes.items()}) | |
| if __name__ == "__main__": | |
| collect() if sys.argv[1] == "collect" else report() | |