Download external_design2code.py from codepawl/loupe: direct link, hf CLI and curl.
- Browser
- Download file 8.76 kB
-
https://huggingface.co/codepawl/loupe/resolve/main/external_design2code.py
- Command line
-
hf download hf://codepawl/loupe/external_design2code.py
-
curl -L -o external_design2code.py https://huggingface.co/codepawl/loupe/resolve/main/external_design2code.py
8.76 kB
| """Loupe on a curated, human-judged benchmark: Design2Code's pairwise human evaluation. | |
| Each row has a reference web page, two pages that models wrote to reproduce it, and the votes of human judges | |
| on which of the two is closer to the reference (SALT-NLP/Design2Code_human_eval_pairwise, 700 rows). Nothing in | |
| Loupe was built or tuned on this data. | |
| Loupe's prediction, fixed before any result was looked at: measure the reference and both candidates in the | |
| reference mode's kinds (text size, weight, spacing above, heading scale, picture shape and width, item shape | |
| and width, input width). A reference check is matched when the candidate has a check of the same kind with the | |
| same name (the same text), and it agrees when the candidate's ratio is within 5 percent. A candidate's score is | |
| the share of the reference's checks that agree; a reference check with no counterpart counts as a miss. The | |
| higher score is predicted closer; scores within 0.02 are a predicted tie. | |
| A plain baseline runs next to it: mean pixel difference between each candidate's screenshot and the | |
| reference's, lower is closer. | |
| Usage: external_design2code.py download | render | report | |
| """ | |
| import json | |
| import subprocess | |
| import sys | |
| from concurrent.futures import ThreadPoolExecutor | |
| from pathlib import Path | |
| sys.stdout.reconfigure(encoding="utf-8", errors="replace") | |
| PACKAGE = Path(__file__).resolve().parent | |
| sys.path.insert(0, str(PACKAGE)) | |
| OUTPUT = PACKAGE / "external" / "design2code_pairwise" | |
| REPOSITORY = "SALT-NLP/Design2Code_human_eval_pairwise" | |
| WIDTH = 1280 | |
| TIE_MARGIN = 0.02 | |
| SPREAD = 1.05 | |
| VARIANTS = ("reference", "candidate_1", "candidate_2") | |
| def rows(): | |
| import pyarrow.parquet as parquet | |
| table = parquet.read_table(OUTPUT / "data.parquet", columns=["id", "ref_html", "model1", "model2", "html1", "html2", "win1", "win2", "tie"]) | |
| return table.to_pylist() | |
| def download(): | |
| from huggingface_hub import hf_hub_download | |
| import shutil | |
| OUTPUT.mkdir(parents=True, exist_ok=True) | |
| shutil.copyfile(hf_hub_download(REPOSITORY, "data/train-00000-of-00001.parquet", repo_type="dataset"), OUTPUT / "data.parquet") | |
| shutil.copyfile(hf_hub_download("SALT-NLP/Design2Code-hf", "rick.jpg", repo_type="dataset"), OUTPUT / "rick.jpg") | |
| found = rows() | |
| print(len(found), "rows; size", (OUTPUT / "data.parquet").stat().st_size // 1024 // 1024, "MiB") | |
| print({key: (value if not isinstance(value, str) or len(value) < 60 else value[:60] + "...") for key, value in found[0].items()}) | |
| print("models", sorted({row["model1"] for row in found} | {row["model2"] for row in found})) | |
| def render_one(job): | |
| index, variant, html = job | |
| directory = OUTPUT / "pages" / f"{index:04d}" / variant | |
| if (directory / "layout.json").exists(): | |
| return "kept" | |
| directory.mkdir(parents=True, exist_ok=True) | |
| (directory / "page.html").write_text(html, encoding="utf-8") | |
| picture = directory / "rick.jpg" | |
| if not picture.exists(): | |
| picture.write_bytes((OUTPUT / "rick.jpg").read_bytes()) | |
| try: | |
| completed = subprocess.run(["node", str(PACKAGE / "collect_page.cjs"), str(directory / "page.html"), str(directory), str(WIDTH), "plain"], | |
| capture_output=True, text=True, encoding="utf-8", timeout=120) | |
| except subprocess.TimeoutExpired: | |
| return "timeout" | |
| return "ok" if completed.returncode == 0 else "failed" | |
| def render(): | |
| jobs = [(index, variant, row[key]) for index, row in enumerate(rows()) | |
| for variant, key in zip(VARIANTS, ("ref_html", "html1", "html2"))] | |
| counts = {} | |
| with ThreadPoolExecutor(max_workers=4) as pool: | |
| for number, status in enumerate(pool.map(render_one, jobs)): | |
| counts[status] = counts.get(status, 0) + 1 | |
| if number % 300 == 0: | |
| print(number, counts, flush=True) | |
| print(counts) | |
| def reference_checks(directory): | |
| """{(kind, name): ratio} for the reference-mode kinds, keeping only names that occur once.""" | |
| import page_audit | |
| found = {} | |
| repeated = set() | |
| for entry in page_audit.measure(directory, "dom"): | |
| if entry["check"] not in page_audit.REFERENCE_KINDS: | |
| continue | |
| key = (entry["check"], entry["target"]) | |
| if key in found: | |
| repeated.add(key) | |
| found[key] = entry["numerator"]["value"] / entry["denominator"]["value"] | |
| return {key: ratio for key, ratio in found.items() if key not in repeated} | |
| def loupe_score(reference, candidate): | |
| if not reference: | |
| return None | |
| agree = sum(key in candidate and reference[key] / SPREAD <= candidate[key] <= reference[key] * SPREAD for key in reference) | |
| return agree / len(reference) | |
| def pixel_distance(directory, reference_directory): | |
| import numpy | |
| from PIL import Image | |
| Image.MAX_IMAGE_PIXELS = None | |
| with Image.open(directory / "page.png") as image: | |
| page = numpy.asarray(image.convert("RGB"), dtype=numpy.float32) | |
| with Image.open(reference_directory / "page.png") as image: | |
| reference = numpy.asarray(image.convert("RGB"), dtype=numpy.float32) | |
| height, width = max(page.shape[0], reference.shape[0]), max(page.shape[1], reference.shape[1]) | |
| def padded(picture): | |
| full = numpy.full((height, width, 3), 255.0, dtype=numpy.float32) | |
| full[:picture.shape[0], :picture.shape[1]] = picture | |
| return full | |
| return float(numpy.abs(padded(page) - padded(reference)).mean() / 255) | |
| def human_label(row): | |
| """1 or 2 for the candidate most judges chose, 0 when the votes do not single one out.""" | |
| if row["win1"] > row["win2"] and row["win1"] > row["tie"]: | |
| return 1 | |
| if row["win2"] > row["win1"] and row["win2"] > row["tie"]: | |
| return 2 | |
| return 0 | |
| def agreement(predictions, labels): | |
| """Accuracy on the rows where the judges chose one candidate, counting a predicted tie as wrong.""" | |
| decided = [(prediction, label) for prediction, label in zip(predictions, labels) if label and prediction is not None] | |
| correct = sum(prediction == label for prediction, label in decided) | |
| ties = sum(prediction == 0 for prediction, _ in decided) | |
| return {"rows": len(decided), "correct": correct, "accuracy": correct / len(decided) if decided else None, "predicted_ties": ties, | |
| "accuracy_when_it_chose": correct / (len(decided) - ties) if len(decided) > ties else None} | |
| def report(): | |
| records = [] | |
| for index, row in enumerate(rows()): | |
| directories = [OUTPUT / "pages" / f"{index:04d}" / variant for variant in VARIANTS] | |
| if not all((directory / "layout.json").exists() for directory in directories): | |
| continue | |
| reference = reference_checks(directories[0]) | |
| scores = [loupe_score(reference, reference_checks(directory)) for directory in directories[1:]] | |
| distances = [pixel_distance(directory, directories[0]) for directory in directories[1:]] | |
| loupe = None if scores[0] is None else 0 if abs(scores[0] - scores[1]) < TIE_MARGIN else 1 if scores[0] > scores[1] else 2 | |
| pixel = 1 if distances[0] < distances[1] else 2 | |
| records.append({"row": index, "id": row["id"], "models": [row["model1"], row["model2"]], "votes": [row["win1"], row["win2"], row["tie"]], | |
| "human": human_label(row), "reference_checks": len(reference), "loupe_scores": scores, "loupe": loupe, | |
| "pixel_distances": distances, "pixel": pixel}) | |
| if index % 100 == 0: | |
| print(index, flush=True) | |
| labels = [record["human"] for record in records] | |
| by_models = {} | |
| for record in records: | |
| if record["human"] and record["loupe"] is not None: | |
| entry = by_models.setdefault(" vs ".join(sorted(record["models"])), {"rows": 0, "loupe_correct": 0, "pixel_correct": 0}) | |
| entry["rows"] += 1 | |
| entry["loupe_correct"] += record["loupe"] == record["human"] | |
| entry["pixel_correct"] += record["pixel"] == record["human"] | |
| summary = {"rows_rendered": len(records), "rows_with_a_human_winner": sum(label != 0 for label in labels), | |
| "rows_without_reference_checks": sum(record["loupe"] is None for record in records), | |
| "loupe": agreement([record["loupe"] for record in records], labels), | |
| "pixel_baseline": agreement([record["pixel"] if record["loupe"] is not None else None for record in records], labels), | |
| "by_model_pair": by_models, "tie_margin": TIE_MARGIN, "spread": SPREAD, "width": WIDTH} | |
| (OUTPUT / "report.json").write_text(json.dumps({**summary, "records": records}, indent=1), encoding="utf-8") | |
| print(json.dumps(summary, indent=1)) | |
| if __name__ == "__main__": | |
| {"download": download, "render": render, "report": report}[sys.argv[1]]() | |