"""Loupe on a curated, human-judged benchmark: Design2Code's pairwise human evaluation. Each row has a reference web page, two pages that models wrote to reproduce it, and the votes of human judges on which of the two is closer to the reference (SALT-NLP/Design2Code_human_eval_pairwise, 700 rows). Nothing in Loupe was built or tuned on this data. Loupe's prediction, fixed before any result was looked at: measure the reference and both candidates in the reference mode's kinds (text size, weight, spacing above, heading scale, picture shape and width, item shape and width, input width). A reference check is matched when the candidate has a check of the same kind with the same name (the same text), and it agrees when the candidate's ratio is within 5 percent. A candidate's score is the share of the reference's checks that agree; a reference check with no counterpart counts as a miss. The higher score is predicted closer; scores within 0.02 are a predicted tie. A plain baseline runs next to it: mean pixel difference between each candidate's screenshot and the reference's, lower is closer. Usage: external_design2code.py download | render | report """ import json import subprocess import sys from concurrent.futures import ThreadPoolExecutor from pathlib import Path sys.stdout.reconfigure(encoding="utf-8", errors="replace") PACKAGE = Path(__file__).resolve().parent sys.path.insert(0, str(PACKAGE)) OUTPUT = PACKAGE / "external" / "design2code_pairwise" REPOSITORY = "SALT-NLP/Design2Code_human_eval_pairwise" WIDTH = 1280 TIE_MARGIN = 0.02 SPREAD = 1.05 VARIANTS = ("reference", "candidate_1", "candidate_2") def rows(): import pyarrow.parquet as parquet table = parquet.read_table(OUTPUT / "data.parquet", columns=["id", "ref_html", "model1", "model2", "html1", "html2", "win1", "win2", "tie"]) return table.to_pylist() def download(): from huggingface_hub import hf_hub_download import shutil OUTPUT.mkdir(parents=True, exist_ok=True) shutil.copyfile(hf_hub_download(REPOSITORY, "data/train-00000-of-00001.parquet", repo_type="dataset"), OUTPUT / "data.parquet") shutil.copyfile(hf_hub_download("SALT-NLP/Design2Code-hf", "rick.jpg", repo_type="dataset"), OUTPUT / "rick.jpg") found = rows() print(len(found), "rows; size", (OUTPUT / "data.parquet").stat().st_size // 1024 // 1024, "MiB") print({key: (value if not isinstance(value, str) or len(value) < 60 else value[:60] + "...") for key, value in found[0].items()}) print("models", sorted({row["model1"] for row in found} | {row["model2"] for row in found})) def render_one(job): index, variant, html = job directory = OUTPUT / "pages" / f"{index:04d}" / variant if (directory / "layout.json").exists(): return "kept" directory.mkdir(parents=True, exist_ok=True) (directory / "page.html").write_text(html, encoding="utf-8") picture = directory / "rick.jpg" if not picture.exists(): picture.write_bytes((OUTPUT / "rick.jpg").read_bytes()) try: completed = subprocess.run(["node", str(PACKAGE / "collect_page.cjs"), str(directory / "page.html"), str(directory), str(WIDTH), "plain"], capture_output=True, text=True, encoding="utf-8", timeout=120) except subprocess.TimeoutExpired: return "timeout" return "ok" if completed.returncode == 0 else "failed" def render(): jobs = [(index, variant, row[key]) for index, row in enumerate(rows()) for variant, key in zip(VARIANTS, ("ref_html", "html1", "html2"))] counts = {} with ThreadPoolExecutor(max_workers=4) as pool: for number, status in enumerate(pool.map(render_one, jobs)): counts[status] = counts.get(status, 0) + 1 if number % 300 == 0: print(number, counts, flush=True) print(counts) def reference_checks(directory): """{(kind, name): ratio} for the reference-mode kinds, keeping only names that occur once.""" import page_audit found = {} repeated = set() for entry in page_audit.measure(directory, "dom"): if entry["check"] not in page_audit.REFERENCE_KINDS: continue key = (entry["check"], entry["target"]) if key in found: repeated.add(key) found[key] = entry["numerator"]["value"] / entry["denominator"]["value"] return {key: ratio for key, ratio in found.items() if key not in repeated} def loupe_score(reference, candidate): if not reference: return None agree = sum(key in candidate and reference[key] / SPREAD <= candidate[key] <= reference[key] * SPREAD for key in reference) return agree / len(reference) def pixel_distance(directory, reference_directory): import numpy from PIL import Image Image.MAX_IMAGE_PIXELS = None with Image.open(directory / "page.png") as image: page = numpy.asarray(image.convert("RGB"), dtype=numpy.float32) with Image.open(reference_directory / "page.png") as image: reference = numpy.asarray(image.convert("RGB"), dtype=numpy.float32) height, width = max(page.shape[0], reference.shape[0]), max(page.shape[1], reference.shape[1]) def padded(picture): full = numpy.full((height, width, 3), 255.0, dtype=numpy.float32) full[:picture.shape[0], :picture.shape[1]] = picture return full return float(numpy.abs(padded(page) - padded(reference)).mean() / 255) def human_label(row): """1 or 2 for the candidate most judges chose, 0 when the votes do not single one out.""" if row["win1"] > row["win2"] and row["win1"] > row["tie"]: return 1 if row["win2"] > row["win1"] and row["win2"] > row["tie"]: return 2 return 0 def agreement(predictions, labels): """Accuracy on the rows where the judges chose one candidate, counting a predicted tie as wrong.""" decided = [(prediction, label) for prediction, label in zip(predictions, labels) if label and prediction is not None] correct = sum(prediction == label for prediction, label in decided) ties = sum(prediction == 0 for prediction, _ in decided) return {"rows": len(decided), "correct": correct, "accuracy": correct / len(decided) if decided else None, "predicted_ties": ties, "accuracy_when_it_chose": correct / (len(decided) - ties) if len(decided) > ties else None} def report(): records = [] for index, row in enumerate(rows()): directories = [OUTPUT / "pages" / f"{index:04d}" / variant for variant in VARIANTS] if not all((directory / "layout.json").exists() for directory in directories): continue reference = reference_checks(directories[0]) scores = [loupe_score(reference, reference_checks(directory)) for directory in directories[1:]] distances = [pixel_distance(directory, directories[0]) for directory in directories[1:]] loupe = None if scores[0] is None else 0 if abs(scores[0] - scores[1]) < TIE_MARGIN else 1 if scores[0] > scores[1] else 2 pixel = 1 if distances[0] < distances[1] else 2 records.append({"row": index, "id": row["id"], "models": [row["model1"], row["model2"]], "votes": [row["win1"], row["win2"], row["tie"]], "human": human_label(row), "reference_checks": len(reference), "loupe_scores": scores, "loupe": loupe, "pixel_distances": distances, "pixel": pixel}) if index % 100 == 0: print(index, flush=True) labels = [record["human"] for record in records] by_models = {} for record in records: if record["human"] and record["loupe"] is not None: entry = by_models.setdefault(" vs ".join(sorted(record["models"])), {"rows": 0, "loupe_correct": 0, "pixel_correct": 0}) entry["rows"] += 1 entry["loupe_correct"] += record["loupe"] == record["human"] entry["pixel_correct"] += record["pixel"] == record["human"] summary = {"rows_rendered": len(records), "rows_with_a_human_winner": sum(label != 0 for label in labels), "rows_without_reference_checks": sum(record["loupe"] is None for record in records), "loupe": agreement([record["loupe"] for record in records], labels), "pixel_baseline": agreement([record["pixel"] if record["loupe"] is not None else None for record in records], labels), "by_model_pair": by_models, "tie_margin": TIE_MARGIN, "spread": SPREAD, "width": WIDTH} (OUTPUT / "report.json").write_text(json.dumps({**summary, "records": records}, indent=1), encoding="utf-8") print(json.dumps(summary, indent=1)) if __name__ == "__main__": {"download": download, "render": render, "report": report}[sys.argv[1]]()