"""Claude Sonnet on the same question as Loupe: which of two copies is closer to the reference page? A seeded sample of the held-out (odd) rows of Design2Code's human pairwise set. For each row Sonnet gets three screenshots made by Loupe's renderer (reference, a, b), scaled to 800 px wide and cut at 2,400 px tall. Which candidate is called "a" is drawn at random per row, so a preference for one letter cannot help. Loupe's choice on the same rows comes from similarity.json's fitted weights. Usage: external_design2code_sonnet.py ask | report """ import json import random import re import subprocess import sys import time from concurrent.futures import ThreadPoolExecutor from pathlib import Path sys.stdout.reconfigure(encoding="utf-8", errors="replace") PACKAGE = Path(__file__).resolve().parent sys.path.insert(0, str(PACKAGE)) import numpy from PIL import Image import similarity Image.MAX_IMAGE_PIXELS = None EXTERNAL = PACKAGE / "external" / "design2code_pairwise" OUTPUT = EXTERNAL / "sonnet" SAMPLE = 100 SEED = 7 PROMPT = ("Open the three image files ref.png, a.png and b.png in the current folder with the Read tool. ref.png is a screenshot of a web page. " "a.png and b.png are screenshots of two attempts to reproduce that page. Which attempt is closer to ref.png in how it looks " "(layout, text, sizes, colours)? Reply with one line of JSON and nothing else: {\"closer\": \"a\"} or {\"closer\": \"b\"}. Do not open any other file.") def sample_rows(): report = json.loads((EXTERNAL / "report.json").read_text(encoding="utf-8")) held_out = [record for record in report["records"] if record["human"] and record["row"] % 2 == 1] chooser = random.Random(SEED) chosen = chooser.sample(held_out, SAMPLE) return [{**record, "first_is_a": chooser.random() < 0.5} for record in chosen] def shrink(source, target): with Image.open(source) as image: picture = image.convert("RGB") picture = picture.resize((800, max(1, round(picture.height * 800 / picture.width)))) picture.crop((0, 0, 800, min(picture.height, 2400))).save(target) def ask_one(record): directory = OUTPUT / f"{record['row']:04d}" if (directory / "answer.txt").exists(): return record["row"], "kept" directory.mkdir(parents=True, exist_ok=True) pages = EXTERNAL / "pages" / f"{record['row']:04d}" first, second = ("a.png", "b.png") if record["first_is_a"] else ("b.png", "a.png") shrink(pages / "reference" / "page.png", directory / "ref.png") shrink(pages / "candidate_1" / "page.png", directory / first) shrink(pages / "candidate_2" / "page.png", directory / second) started = time.perf_counter() completed = subprocess.run(["claude.cmd", "-p", "--model", "sonnet", "--allowedTools", "Read", "--strict-mcp-config"], input=PROMPT, capture_output=True, text=True, encoding="utf-8", cwd=str(directory), timeout=600) if completed.returncode != 0 or not completed.stdout.strip(): return record["row"], "failed" (directory / "answer.txt").write_text(completed.stdout, encoding="utf-8") (directory / "seconds.txt").write_text(f"{time.perf_counter() - started:.1f}", encoding="utf-8") return record["row"], "ok" def ask(): OUTPUT.mkdir(parents=True, exist_ok=True) (OUTPUT / "prompt.txt").write_text(PROMPT, encoding="utf-8") counts = {} with ThreadPoolExecutor(max_workers=3) as pool: for _, status in pool.map(ask_one, sample_rows()): counts[status] = counts.get(status, 0) + 1 print(counts) def loupe_choice(record, model): directory = EXTERNAL / "pages" / f"{record['row']:04d}" layouts = [json.loads((directory / variant / "layout.json").read_text(encoding="utf-8")) for variant in ("reference", "candidate_1", "candidate_2")] first, second = similarity.distances(layouts[0], layouts[1]), similarity.distances(layouts[0], layouts[2]) difference = numpy.asarray([second[name] - first[name] for name in similarity.FEATURES]) / numpy.asarray(model["scale"]) return 1 if float(difference @ numpy.asarray(model["weights"])) > 0 else 2 def report(): model = json.loads((PACKAGE / "similarity.json").read_text(encoding="utf-8"))["models"]["loupe_distances"] rows = [] for record in sample_rows(): answer_path = OUTPUT / f"{record['row']:04d}" / "answer.txt" if not answer_path.exists(): continue match = re.search(r"\"closer\"\s*:\s*\"([ab])\"", answer_path.read_text(encoding="utf-8")) if match is None: continue said_a = match.group(1) == "a" sonnet = 1 if said_a == record["first_is_a"] else 2 rows.append({"row": record["row"], "human": record["human"], "sonnet": sonnet, "loupe": loupe_choice(record, model), "pixel": record["pixel"], "seconds": float((OUTPUT / f"{record['row']:04d}" / "seconds.txt").read_text(encoding="utf-8"))}) share = lambda key: sum(row[key] == row["human"] for row in rows) / len(rows) summary = {"rows": len(rows), "sample": SAMPLE, "seed": SEED, "sonnet_agreement": share("sonnet"), "loupe_agreement": share("loupe"), "pixel_agreement": share("pixel"), "sonnet_seconds_median": sorted(row["seconds"] for row in rows)[len(rows) // 2], "sonnet_and_loupe_agree": sum(row["sonnet"] == row["loupe"] for row in rows) / len(rows)} (OUTPUT / "report.json").write_text(json.dumps({**summary, "rows_detail": rows}, indent=1), encoding="utf-8") print(summary) if __name__ == "__main__": ask() if sys.argv[1] == "ask" else report()