File size: 8,761 Bytes
e57bf63 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 | """Loupe on a curated, human-judged benchmark: Design2Code's pairwise human evaluation.
Each row has a reference web page, two pages that models wrote to reproduce it, and the votes of human judges
on which of the two is closer to the reference (SALT-NLP/Design2Code_human_eval_pairwise, 700 rows). Nothing in
Loupe was built or tuned on this data.
Loupe's prediction, fixed before any result was looked at: measure the reference and both candidates in the
reference mode's kinds (text size, weight, spacing above, heading scale, picture shape and width, item shape
and width, input width). A reference check is matched when the candidate has a check of the same kind with the
same name (the same text), and it agrees when the candidate's ratio is within 5 percent. A candidate's score is
the share of the reference's checks that agree; a reference check with no counterpart counts as a miss. The
higher score is predicted closer; scores within 0.02 are a predicted tie.
A plain baseline runs next to it: mean pixel difference between each candidate's screenshot and the
reference's, lower is closer.
Usage: external_design2code.py download | render | report
"""
import json
import subprocess
import sys
from concurrent.futures import ThreadPoolExecutor
from pathlib import Path
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
PACKAGE = Path(__file__).resolve().parent
sys.path.insert(0, str(PACKAGE))
OUTPUT = PACKAGE / "external" / "design2code_pairwise"
REPOSITORY = "SALT-NLP/Design2Code_human_eval_pairwise"
WIDTH = 1280
TIE_MARGIN = 0.02
SPREAD = 1.05
VARIANTS = ("reference", "candidate_1", "candidate_2")
def rows():
import pyarrow.parquet as parquet
table = parquet.read_table(OUTPUT / "data.parquet", columns=["id", "ref_html", "model1", "model2", "html1", "html2", "win1", "win2", "tie"])
return table.to_pylist()
def download():
from huggingface_hub import hf_hub_download
import shutil
OUTPUT.mkdir(parents=True, exist_ok=True)
shutil.copyfile(hf_hub_download(REPOSITORY, "data/train-00000-of-00001.parquet", repo_type="dataset"), OUTPUT / "data.parquet")
shutil.copyfile(hf_hub_download("SALT-NLP/Design2Code-hf", "rick.jpg", repo_type="dataset"), OUTPUT / "rick.jpg")
found = rows()
print(len(found), "rows; size", (OUTPUT / "data.parquet").stat().st_size // 1024 // 1024, "MiB")
print({key: (value if not isinstance(value, str) or len(value) < 60 else value[:60] + "...") for key, value in found[0].items()})
print("models", sorted({row["model1"] for row in found} | {row["model2"] for row in found}))
def render_one(job):
index, variant, html = job
directory = OUTPUT / "pages" / f"{index:04d}" / variant
if (directory / "layout.json").exists():
return "kept"
directory.mkdir(parents=True, exist_ok=True)
(directory / "page.html").write_text(html, encoding="utf-8")
picture = directory / "rick.jpg"
if not picture.exists():
picture.write_bytes((OUTPUT / "rick.jpg").read_bytes())
try:
completed = subprocess.run(["node", str(PACKAGE / "collect_page.cjs"), str(directory / "page.html"), str(directory), str(WIDTH), "plain"],
capture_output=True, text=True, encoding="utf-8", timeout=120)
except subprocess.TimeoutExpired:
return "timeout"
return "ok" if completed.returncode == 0 else "failed"
def render():
jobs = [(index, variant, row[key]) for index, row in enumerate(rows())
for variant, key in zip(VARIANTS, ("ref_html", "html1", "html2"))]
counts = {}
with ThreadPoolExecutor(max_workers=4) as pool:
for number, status in enumerate(pool.map(render_one, jobs)):
counts[status] = counts.get(status, 0) + 1
if number % 300 == 0:
print(number, counts, flush=True)
print(counts)
def reference_checks(directory):
"""{(kind, name): ratio} for the reference-mode kinds, keeping only names that occur once."""
import page_audit
found = {}
repeated = set()
for entry in page_audit.measure(directory, "dom"):
if entry["check"] not in page_audit.REFERENCE_KINDS:
continue
key = (entry["check"], entry["target"])
if key in found:
repeated.add(key)
found[key] = entry["numerator"]["value"] / entry["denominator"]["value"]
return {key: ratio for key, ratio in found.items() if key not in repeated}
def loupe_score(reference, candidate):
if not reference:
return None
agree = sum(key in candidate and reference[key] / SPREAD <= candidate[key] <= reference[key] * SPREAD for key in reference)
return agree / len(reference)
def pixel_distance(directory, reference_directory):
import numpy
from PIL import Image
Image.MAX_IMAGE_PIXELS = None
with Image.open(directory / "page.png") as image:
page = numpy.asarray(image.convert("RGB"), dtype=numpy.float32)
with Image.open(reference_directory / "page.png") as image:
reference = numpy.asarray(image.convert("RGB"), dtype=numpy.float32)
height, width = max(page.shape[0], reference.shape[0]), max(page.shape[1], reference.shape[1])
def padded(picture):
full = numpy.full((height, width, 3), 255.0, dtype=numpy.float32)
full[:picture.shape[0], :picture.shape[1]] = picture
return full
return float(numpy.abs(padded(page) - padded(reference)).mean() / 255)
def human_label(row):
"""1 or 2 for the candidate most judges chose, 0 when the votes do not single one out."""
if row["win1"] > row["win2"] and row["win1"] > row["tie"]:
return 1
if row["win2"] > row["win1"] and row["win2"] > row["tie"]:
return 2
return 0
def agreement(predictions, labels):
"""Accuracy on the rows where the judges chose one candidate, counting a predicted tie as wrong."""
decided = [(prediction, label) for prediction, label in zip(predictions, labels) if label and prediction is not None]
correct = sum(prediction == label for prediction, label in decided)
ties = sum(prediction == 0 for prediction, _ in decided)
return {"rows": len(decided), "correct": correct, "accuracy": correct / len(decided) if decided else None, "predicted_ties": ties,
"accuracy_when_it_chose": correct / (len(decided) - ties) if len(decided) > ties else None}
def report():
records = []
for index, row in enumerate(rows()):
directories = [OUTPUT / "pages" / f"{index:04d}" / variant for variant in VARIANTS]
if not all((directory / "layout.json").exists() for directory in directories):
continue
reference = reference_checks(directories[0])
scores = [loupe_score(reference, reference_checks(directory)) for directory in directories[1:]]
distances = [pixel_distance(directory, directories[0]) for directory in directories[1:]]
loupe = None if scores[0] is None else 0 if abs(scores[0] - scores[1]) < TIE_MARGIN else 1 if scores[0] > scores[1] else 2
pixel = 1 if distances[0] < distances[1] else 2
records.append({"row": index, "id": row["id"], "models": [row["model1"], row["model2"]], "votes": [row["win1"], row["win2"], row["tie"]],
"human": human_label(row), "reference_checks": len(reference), "loupe_scores": scores, "loupe": loupe,
"pixel_distances": distances, "pixel": pixel})
if index % 100 == 0:
print(index, flush=True)
labels = [record["human"] for record in records]
by_models = {}
for record in records:
if record["human"] and record["loupe"] is not None:
entry = by_models.setdefault(" vs ".join(sorted(record["models"])), {"rows": 0, "loupe_correct": 0, "pixel_correct": 0})
entry["rows"] += 1
entry["loupe_correct"] += record["loupe"] == record["human"]
entry["pixel_correct"] += record["pixel"] == record["human"]
summary = {"rows_rendered": len(records), "rows_with_a_human_winner": sum(label != 0 for label in labels),
"rows_without_reference_checks": sum(record["loupe"] is None for record in records),
"loupe": agreement([record["loupe"] for record in records], labels),
"pixel_baseline": agreement([record["pixel"] if record["loupe"] is not None else None for record in records], labels),
"by_model_pair": by_models, "tie_margin": TIE_MARGIN, "spread": SPREAD, "width": WIDTH}
(OUTPUT / "report.json").write_text(json.dumps({**summary, "records": records}, indent=1), encoding="utf-8")
print(json.dumps(summary, indent=1))
if __name__ == "__main__":
{"download": download, "render": render, "report": report}[sys.argv[1]]()
|