File size: 8,761 Bytes
e57bf63
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
"""Loupe on a curated, human-judged benchmark: Design2Code's pairwise human evaluation.

Each row has a reference web page, two pages that models wrote to reproduce it, and the votes of human judges
on which of the two is closer to the reference (SALT-NLP/Design2Code_human_eval_pairwise, 700 rows). Nothing in
Loupe was built or tuned on this data.

Loupe's prediction, fixed before any result was looked at: measure the reference and both candidates in the
reference mode's kinds (text size, weight, spacing above, heading scale, picture shape and width, item shape
and width, input width). A reference check is matched when the candidate has a check of the same kind with the
same name (the same text), and it agrees when the candidate's ratio is within 5 percent. A candidate's score is
the share of the reference's checks that agree; a reference check with no counterpart counts as a miss. The
higher score is predicted closer; scores within 0.02 are a predicted tie.

A plain baseline runs next to it: mean pixel difference between each candidate's screenshot and the
reference's, lower is closer.

Usage: external_design2code.py download | render | report
"""

import json
import subprocess
import sys
from concurrent.futures import ThreadPoolExecutor
from pathlib import Path

sys.stdout.reconfigure(encoding="utf-8", errors="replace")
PACKAGE = Path(__file__).resolve().parent
sys.path.insert(0, str(PACKAGE))

OUTPUT = PACKAGE / "external" / "design2code_pairwise"
REPOSITORY = "SALT-NLP/Design2Code_human_eval_pairwise"
WIDTH = 1280
TIE_MARGIN = 0.02
SPREAD = 1.05
VARIANTS = ("reference", "candidate_1", "candidate_2")


def rows():
    import pyarrow.parquet as parquet

    table = parquet.read_table(OUTPUT / "data.parquet", columns=["id", "ref_html", "model1", "model2", "html1", "html2", "win1", "win2", "tie"])
    return table.to_pylist()


def download():
    from huggingface_hub import hf_hub_download
    import shutil

    OUTPUT.mkdir(parents=True, exist_ok=True)
    shutil.copyfile(hf_hub_download(REPOSITORY, "data/train-00000-of-00001.parquet", repo_type="dataset"), OUTPUT / "data.parquet")
    shutil.copyfile(hf_hub_download("SALT-NLP/Design2Code-hf", "rick.jpg", repo_type="dataset"), OUTPUT / "rick.jpg")
    found = rows()
    print(len(found), "rows; size", (OUTPUT / "data.parquet").stat().st_size // 1024 // 1024, "MiB")
    print({key: (value if not isinstance(value, str) or len(value) < 60 else value[:60] + "...") for key, value in found[0].items()})
    print("models", sorted({row["model1"] for row in found} | {row["model2"] for row in found}))


def render_one(job):
    index, variant, html = job
    directory = OUTPUT / "pages" / f"{index:04d}" / variant
    if (directory / "layout.json").exists():
        return "kept"
    directory.mkdir(parents=True, exist_ok=True)
    (directory / "page.html").write_text(html, encoding="utf-8")
    picture = directory / "rick.jpg"
    if not picture.exists():
        picture.write_bytes((OUTPUT / "rick.jpg").read_bytes())
    try:
        completed = subprocess.run(["node", str(PACKAGE / "collect_page.cjs"), str(directory / "page.html"), str(directory), str(WIDTH), "plain"],
                                   capture_output=True, text=True, encoding="utf-8", timeout=120)
    except subprocess.TimeoutExpired:
        return "timeout"
    return "ok" if completed.returncode == 0 else "failed"


def render():
    jobs = [(index, variant, row[key]) for index, row in enumerate(rows())
            for variant, key in zip(VARIANTS, ("ref_html", "html1", "html2"))]
    counts = {}
    with ThreadPoolExecutor(max_workers=4) as pool:
        for number, status in enumerate(pool.map(render_one, jobs)):
            counts[status] = counts.get(status, 0) + 1
            if number % 300 == 0:
                print(number, counts, flush=True)
    print(counts)


def reference_checks(directory):
    """{(kind, name): ratio} for the reference-mode kinds, keeping only names that occur once."""
    import page_audit

    found = {}
    repeated = set()
    for entry in page_audit.measure(directory, "dom"):
        if entry["check"] not in page_audit.REFERENCE_KINDS:
            continue
        key = (entry["check"], entry["target"])
        if key in found:
            repeated.add(key)
        found[key] = entry["numerator"]["value"] / entry["denominator"]["value"]
    return {key: ratio for key, ratio in found.items() if key not in repeated}


def loupe_score(reference, candidate):
    if not reference:
        return None
    agree = sum(key in candidate and reference[key] / SPREAD <= candidate[key] <= reference[key] * SPREAD for key in reference)
    return agree / len(reference)


def pixel_distance(directory, reference_directory):
    import numpy
    from PIL import Image

    Image.MAX_IMAGE_PIXELS = None
    with Image.open(directory / "page.png") as image:
        page = numpy.asarray(image.convert("RGB"), dtype=numpy.float32)
    with Image.open(reference_directory / "page.png") as image:
        reference = numpy.asarray(image.convert("RGB"), dtype=numpy.float32)
    height, width = max(page.shape[0], reference.shape[0]), max(page.shape[1], reference.shape[1])

    def padded(picture):
        full = numpy.full((height, width, 3), 255.0, dtype=numpy.float32)
        full[:picture.shape[0], :picture.shape[1]] = picture
        return full

    return float(numpy.abs(padded(page) - padded(reference)).mean() / 255)


def human_label(row):
    """1 or 2 for the candidate most judges chose, 0 when the votes do not single one out."""
    if row["win1"] > row["win2"] and row["win1"] > row["tie"]:
        return 1
    if row["win2"] > row["win1"] and row["win2"] > row["tie"]:
        return 2
    return 0


def agreement(predictions, labels):
    """Accuracy on the rows where the judges chose one candidate, counting a predicted tie as wrong."""
    decided = [(prediction, label) for prediction, label in zip(predictions, labels) if label and prediction is not None]
    correct = sum(prediction == label for prediction, label in decided)
    ties = sum(prediction == 0 for prediction, _ in decided)
    return {"rows": len(decided), "correct": correct, "accuracy": correct / len(decided) if decided else None, "predicted_ties": ties,
            "accuracy_when_it_chose": correct / (len(decided) - ties) if len(decided) > ties else None}


def report():
    records = []
    for index, row in enumerate(rows()):
        directories = [OUTPUT / "pages" / f"{index:04d}" / variant for variant in VARIANTS]
        if not all((directory / "layout.json").exists() for directory in directories):
            continue
        reference = reference_checks(directories[0])
        scores = [loupe_score(reference, reference_checks(directory)) for directory in directories[1:]]
        distances = [pixel_distance(directory, directories[0]) for directory in directories[1:]]
        loupe = None if scores[0] is None else 0 if abs(scores[0] - scores[1]) < TIE_MARGIN else 1 if scores[0] > scores[1] else 2
        pixel = 1 if distances[0] < distances[1] else 2
        records.append({"row": index, "id": row["id"], "models": [row["model1"], row["model2"]], "votes": [row["win1"], row["win2"], row["tie"]],
                        "human": human_label(row), "reference_checks": len(reference), "loupe_scores": scores, "loupe": loupe,
                        "pixel_distances": distances, "pixel": pixel})
        if index % 100 == 0:
            print(index, flush=True)
    labels = [record["human"] for record in records]
    by_models = {}
    for record in records:
        if record["human"] and record["loupe"] is not None:
            entry = by_models.setdefault(" vs ".join(sorted(record["models"])), {"rows": 0, "loupe_correct": 0, "pixel_correct": 0})
            entry["rows"] += 1
            entry["loupe_correct"] += record["loupe"] == record["human"]
            entry["pixel_correct"] += record["pixel"] == record["human"]
    summary = {"rows_rendered": len(records), "rows_with_a_human_winner": sum(label != 0 for label in labels),
               "rows_without_reference_checks": sum(record["loupe"] is None for record in records),
               "loupe": agreement([record["loupe"] for record in records], labels),
               "pixel_baseline": agreement([record["pixel"] if record["loupe"] is not None else None for record in records], labels),
               "by_model_pair": by_models, "tie_margin": TIE_MARGIN, "spread": SPREAD, "width": WIDTH}
    (OUTPUT / "report.json").write_text(json.dumps({**summary, "records": records}, indent=1), encoding="utf-8")
    print(json.dumps(summary, indent=1))


if __name__ == "__main__":
    {"download": download, "render": render, "report": report}[sys.argv[1]]()