loupe / external_design2code.py
anxuanng's picture
Add the Design2Code human-judged benchmark: Loupe does not beat a pixel baseline there
e57bf63 verified
Raw History Blame Contribute Delete
8.76 kB
"""Loupe on a curated, human-judged benchmark: Design2Code's pairwise human evaluation.
Each row has a reference web page, two pages that models wrote to reproduce it, and the votes of human judges
on which of the two is closer to the reference (SALT-NLP/Design2Code_human_eval_pairwise, 700 rows). Nothing in
Loupe was built or tuned on this data.
Loupe's prediction, fixed before any result was looked at: measure the reference and both candidates in the
reference mode's kinds (text size, weight, spacing above, heading scale, picture shape and width, item shape
and width, input width). A reference check is matched when the candidate has a check of the same kind with the
same name (the same text), and it agrees when the candidate's ratio is within 5 percent. A candidate's score is
the share of the reference's checks that agree; a reference check with no counterpart counts as a miss. The
higher score is predicted closer; scores within 0.02 are a predicted tie.
A plain baseline runs next to it: mean pixel difference between each candidate's screenshot and the
reference's, lower is closer.
Usage: external_design2code.py download | render | report
"""
import json
import subprocess
import sys
from concurrent.futures import ThreadPoolExecutor
from pathlib import Path
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
PACKAGE = Path(__file__).resolve().parent
sys.path.insert(0, str(PACKAGE))
OUTPUT = PACKAGE / "external" / "design2code_pairwise"
REPOSITORY = "SALT-NLP/Design2Code_human_eval_pairwise"
WIDTH = 1280
TIE_MARGIN = 0.02
SPREAD = 1.05
VARIANTS = ("reference", "candidate_1", "candidate_2")
def rows():
import pyarrow.parquet as parquet
table = parquet.read_table(OUTPUT / "data.parquet", columns=["id", "ref_html", "model1", "model2", "html1", "html2", "win1", "win2", "tie"])
return table.to_pylist()
def download():
from huggingface_hub import hf_hub_download
import shutil
OUTPUT.mkdir(parents=True, exist_ok=True)
shutil.copyfile(hf_hub_download(REPOSITORY, "data/train-00000-of-00001.parquet", repo_type="dataset"), OUTPUT / "data.parquet")
shutil.copyfile(hf_hub_download("SALT-NLP/Design2Code-hf", "rick.jpg", repo_type="dataset"), OUTPUT / "rick.jpg")
found = rows()
print(len(found), "rows; size", (OUTPUT / "data.parquet").stat().st_size // 1024 // 1024, "MiB")
print({key: (value if not isinstance(value, str) or len(value) < 60 else value[:60] + "...") for key, value in found[0].items()})
print("models", sorted({row["model1"] for row in found} | {row["model2"] for row in found}))
def render_one(job):
index, variant, html = job
directory = OUTPUT / "pages" / f"{index:04d}" / variant
if (directory / "layout.json").exists():
return "kept"
directory.mkdir(parents=True, exist_ok=True)
(directory / "page.html").write_text(html, encoding="utf-8")
picture = directory / "rick.jpg"
if not picture.exists():
picture.write_bytes((OUTPUT / "rick.jpg").read_bytes())
try:
completed = subprocess.run(["node", str(PACKAGE / "collect_page.cjs"), str(directory / "page.html"), str(directory), str(WIDTH), "plain"],
capture_output=True, text=True, encoding="utf-8", timeout=120)
except subprocess.TimeoutExpired:
return "timeout"
return "ok" if completed.returncode == 0 else "failed"
def render():
jobs = [(index, variant, row[key]) for index, row in enumerate(rows())
for variant, key in zip(VARIANTS, ("ref_html", "html1", "html2"))]
counts = {}
with ThreadPoolExecutor(max_workers=4) as pool:
for number, status in enumerate(pool.map(render_one, jobs)):
counts[status] = counts.get(status, 0) + 1
if number % 300 == 0:
print(number, counts, flush=True)
print(counts)
def reference_checks(directory):
"""{(kind, name): ratio} for the reference-mode kinds, keeping only names that occur once."""
import page_audit
found = {}
repeated = set()
for entry in page_audit.measure(directory, "dom"):
if entry["check"] not in page_audit.REFERENCE_KINDS:
continue
key = (entry["check"], entry["target"])
if key in found:
repeated.add(key)
found[key] = entry["numerator"]["value"] / entry["denominator"]["value"]
return {key: ratio for key, ratio in found.items() if key not in repeated}
def loupe_score(reference, candidate):
if not reference:
return None
agree = sum(key in candidate and reference[key] / SPREAD <= candidate[key] <= reference[key] * SPREAD for key in reference)
return agree / len(reference)
def pixel_distance(directory, reference_directory):
import numpy
from PIL import Image
Image.MAX_IMAGE_PIXELS = None
with Image.open(directory / "page.png") as image:
page = numpy.asarray(image.convert("RGB"), dtype=numpy.float32)
with Image.open(reference_directory / "page.png") as image:
reference = numpy.asarray(image.convert("RGB"), dtype=numpy.float32)
height, width = max(page.shape[0], reference.shape[0]), max(page.shape[1], reference.shape[1])
def padded(picture):
full = numpy.full((height, width, 3), 255.0, dtype=numpy.float32)
full[:picture.shape[0], :picture.shape[1]] = picture
return full
return float(numpy.abs(padded(page) - padded(reference)).mean() / 255)
def human_label(row):
"""1 or 2 for the candidate most judges chose, 0 when the votes do not single one out."""
if row["win1"] > row["win2"] and row["win1"] > row["tie"]:
return 1
if row["win2"] > row["win1"] and row["win2"] > row["tie"]:
return 2
return 0
def agreement(predictions, labels):
"""Accuracy on the rows where the judges chose one candidate, counting a predicted tie as wrong."""
decided = [(prediction, label) for prediction, label in zip(predictions, labels) if label and prediction is not None]
correct = sum(prediction == label for prediction, label in decided)
ties = sum(prediction == 0 for prediction, _ in decided)
return {"rows": len(decided), "correct": correct, "accuracy": correct / len(decided) if decided else None, "predicted_ties": ties,
"accuracy_when_it_chose": correct / (len(decided) - ties) if len(decided) > ties else None}
def report():
records = []
for index, row in enumerate(rows()):
directories = [OUTPUT / "pages" / f"{index:04d}" / variant for variant in VARIANTS]
if not all((directory / "layout.json").exists() for directory in directories):
continue
reference = reference_checks(directories[0])
scores = [loupe_score(reference, reference_checks(directory)) for directory in directories[1:]]
distances = [pixel_distance(directory, directories[0]) for directory in directories[1:]]
loupe = None if scores[0] is None else 0 if abs(scores[0] - scores[1]) < TIE_MARGIN else 1 if scores[0] > scores[1] else 2
pixel = 1 if distances[0] < distances[1] else 2
records.append({"row": index, "id": row["id"], "models": [row["model1"], row["model2"]], "votes": [row["win1"], row["win2"], row["tie"]],
"human": human_label(row), "reference_checks": len(reference), "loupe_scores": scores, "loupe": loupe,
"pixel_distances": distances, "pixel": pixel})
if index % 100 == 0:
print(index, flush=True)
labels = [record["human"] for record in records]
by_models = {}
for record in records:
if record["human"] and record["loupe"] is not None:
entry = by_models.setdefault(" vs ".join(sorted(record["models"])), {"rows": 0, "loupe_correct": 0, "pixel_correct": 0})
entry["rows"] += 1
entry["loupe_correct"] += record["loupe"] == record["human"]
entry["pixel_correct"] += record["pixel"] == record["human"]
summary = {"rows_rendered": len(records), "rows_with_a_human_winner": sum(label != 0 for label in labels),
"rows_without_reference_checks": sum(record["loupe"] is None for record in records),
"loupe": agreement([record["loupe"] for record in records], labels),
"pixel_baseline": agreement([record["pixel"] if record["loupe"] is not None else None for record in records], labels),
"by_model_pair": by_models, "tie_margin": TIE_MARGIN, "spread": SPREAD, "width": WIDTH}
(OUTPUT / "report.json").write_text(json.dumps({**summary, "records": records}, indent=1), encoding="utf-8")
print(json.dumps(summary, indent=1))
if __name__ == "__main__":
{"download": download, "render": render, "report": report}[sys.argv[1]]()