File size: 5,632 Bytes
5fd9805
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
"""Claude Sonnet on the same question as Loupe: which of two copies is closer to the reference page?

A seeded sample of the held-out (odd) rows of Design2Code's human pairwise set. For each row Sonnet gets three
screenshots made by Loupe's renderer (reference, a, b), scaled to 800 px wide and cut at 2,400 px tall. Which
candidate is called "a" is drawn at random per row, so a preference for one letter cannot help. Loupe's choice
on the same rows comes from similarity.json's fitted weights.

Usage: external_design2code_sonnet.py ask | report
"""

import json
import random
import re
import subprocess
import sys
import time
from concurrent.futures import ThreadPoolExecutor
from pathlib import Path

sys.stdout.reconfigure(encoding="utf-8", errors="replace")
PACKAGE = Path(__file__).resolve().parent
sys.path.insert(0, str(PACKAGE))

import numpy
from PIL import Image
import similarity

Image.MAX_IMAGE_PIXELS = None
EXTERNAL = PACKAGE / "external" / "design2code_pairwise"
OUTPUT = EXTERNAL / "sonnet"
SAMPLE = 100
SEED = 7
PROMPT = ("Open the three image files ref.png, a.png and b.png in the current folder with the Read tool. ref.png is a screenshot of a web page. "
          "a.png and b.png are screenshots of two attempts to reproduce that page. Which attempt is closer to ref.png in how it looks "
          "(layout, text, sizes, colours)? Reply with one line of JSON and nothing else: {\"closer\": \"a\"} or {\"closer\": \"b\"}. Do not open any other file.")


def sample_rows():
    report = json.loads((EXTERNAL / "report.json").read_text(encoding="utf-8"))
    held_out = [record for record in report["records"] if record["human"] and record["row"] % 2 == 1]
    chooser = random.Random(SEED)
    chosen = chooser.sample(held_out, SAMPLE)
    return [{**record, "first_is_a": chooser.random() < 0.5} for record in chosen]


def shrink(source, target):
    with Image.open(source) as image:
        picture = image.convert("RGB")
        picture = picture.resize((800, max(1, round(picture.height * 800 / picture.width))))
        picture.crop((0, 0, 800, min(picture.height, 2400))).save(target)


def ask_one(record):
    directory = OUTPUT / f"{record['row']:04d}"
    if (directory / "answer.txt").exists():
        return record["row"], "kept"
    directory.mkdir(parents=True, exist_ok=True)
    pages = EXTERNAL / "pages" / f"{record['row']:04d}"
    first, second = ("a.png", "b.png") if record["first_is_a"] else ("b.png", "a.png")
    shrink(pages / "reference" / "page.png", directory / "ref.png")
    shrink(pages / "candidate_1" / "page.png", directory / first)
    shrink(pages / "candidate_2" / "page.png", directory / second)
    started = time.perf_counter()
    completed = subprocess.run(["claude.cmd", "-p", "--model", "sonnet", "--allowedTools", "Read", "--strict-mcp-config"], input=PROMPT,
                               capture_output=True, text=True, encoding="utf-8", cwd=str(directory), timeout=600)
    if completed.returncode != 0 or not completed.stdout.strip():
        return record["row"], "failed"
    (directory / "answer.txt").write_text(completed.stdout, encoding="utf-8")
    (directory / "seconds.txt").write_text(f"{time.perf_counter() - started:.1f}", encoding="utf-8")
    return record["row"], "ok"


def ask():
    OUTPUT.mkdir(parents=True, exist_ok=True)
    (OUTPUT / "prompt.txt").write_text(PROMPT, encoding="utf-8")
    counts = {}
    with ThreadPoolExecutor(max_workers=3) as pool:
        for _, status in pool.map(ask_one, sample_rows()):
            counts[status] = counts.get(status, 0) + 1
    print(counts)


def loupe_choice(record, model):
    directory = EXTERNAL / "pages" / f"{record['row']:04d}"
    layouts = [json.loads((directory / variant / "layout.json").read_text(encoding="utf-8")) for variant in ("reference", "candidate_1", "candidate_2")]
    first, second = similarity.distances(layouts[0], layouts[1]), similarity.distances(layouts[0], layouts[2])
    difference = numpy.asarray([second[name] - first[name] for name in similarity.FEATURES]) / numpy.asarray(model["scale"])
    return 1 if float(difference @ numpy.asarray(model["weights"])) > 0 else 2


def report():
    model = json.loads((PACKAGE / "similarity.json").read_text(encoding="utf-8"))["models"]["loupe_distances"]
    rows = []
    for record in sample_rows():
        answer_path = OUTPUT / f"{record['row']:04d}" / "answer.txt"
        if not answer_path.exists():
            continue
        match = re.search(r"\"closer\"\s*:\s*\"([ab])\"", answer_path.read_text(encoding="utf-8"))
        if match is None:
            continue
        said_a = match.group(1) == "a"
        sonnet = 1 if said_a == record["first_is_a"] else 2
        rows.append({"row": record["row"], "human": record["human"], "sonnet": sonnet, "loupe": loupe_choice(record, model), "pixel": record["pixel"],
                     "seconds": float((OUTPUT / f"{record['row']:04d}" / "seconds.txt").read_text(encoding="utf-8"))})
    share = lambda key: sum(row[key] == row["human"] for row in rows) / len(rows)
    summary = {"rows": len(rows), "sample": SAMPLE, "seed": SEED, "sonnet_agreement": share("sonnet"), "loupe_agreement": share("loupe"),
               "pixel_agreement": share("pixel"), "sonnet_seconds_median": sorted(row["seconds"] for row in rows)[len(rows) // 2],
               "sonnet_and_loupe_agree": sum(row["sonnet"] == row["loupe"] for row in rows) / len(rows)}
    (OUTPUT / "report.json").write_text(json.dumps({**summary, "rows_detail": rows}, indent=1), encoding="utf-8")
    print(summary)


if __name__ == "__main__":
    ask() if sys.argv[1] == "ask" else report()