Download scripts/build_viewer.py from timchen0618/structure-generation-viewer: direct link, hf CLI and curl.
- Browser
- Download file 14.8 kB
-
https://huggingface.co/spaces/timchen0618/structure-generation-viewer/resolve/main/scripts/build_viewer.py
- Command line
-
hf download hf://spaces/timchen0618/structure-generation-viewer/scripts/build_viewer.py
-
curl -L -o build_viewer.py https://huggingface.co/spaces/timchen0618/structure-generation-viewer/resolve/main/scripts/build_viewer.py
14.8 kB
| #!/usr/bin/env python3 | |
| """Build a static cross-dataset source-document and scaffold viewer.""" | |
| from __future__ import annotations | |
| import argparse | |
| import csv | |
| import hashlib | |
| import json | |
| import shutil | |
| from pathlib import Path | |
| from typing import Any, Iterable | |
| ROOT = Path(__file__).resolve().parents[1] | |
| INFO = Path("/home/azureuser/projects/information-scaffolds") | |
| SCHEMA_REPAIR = Path("/home/azureuser/projects/information-scaffolds-schema-repair") | |
| SESSION_INPUTS = Path( | |
| "/home/azureuser/.copilot/session-state/6a2da107-0228-4316-ad25-f248cf796604/" | |
| "files/structure-generation-viewer-inputs" | |
| ) | |
| DATA_ROOT = Path("/mnt/ramdisk/blobstore/timchen0618/data") | |
| SHAPE_ALIASES = { | |
| "tabular_records": "tables", | |
| "tables": "tables", | |
| "relation_graphs_and_mappings": "knowledge_graphs", | |
| "knowledge_graphs": "knowledge_graphs", | |
| "claim_and_theme_summaries": "claims", | |
| "claims": "claims", | |
| "chronology_and_timeline_indexes": "timelines", | |
| "timelines": "timelines", | |
| "qa_shortcuts_and_templates": "legacy_qa_shortcuts", | |
| } | |
| SHAPE_LABELS = { | |
| "tables": "Tables", | |
| "knowledge_graphs": "Knowledge graphs", | |
| "claims": "Claims", | |
| "timelines": "Timelines", | |
| "legacy_qa_shortcuts": "Legacy QA shortcuts", | |
| } | |
| DATASETS = { | |
| "monaco_dev": { | |
| "label": "MoNaCo-dev", | |
| "source": { | |
| "kind": "raw_dir", | |
| "path": INFO / "outputs/rawtext_corpus_monacodev/monaco_dev/scaffolds", | |
| }, | |
| "runs": [ | |
| { | |
| "id": "legacy_full", | |
| "label": "Legacy full extraction - 50-doc subset", | |
| "short_label": "Legacy", | |
| "job": "sweet_screw_frnspf41x5", | |
| "schema": "legacy five-shape", | |
| "root": INFO | |
| / "outputs/e2e_dev_scaffolds/monaco_dev/named-outputs/scaffolds_dir", | |
| }, | |
| { | |
| "id": "current", | |
| "label": "Current schema-v3 extraction", | |
| "short_label": "Current", | |
| "job": "quirky_cabbage_zdpp543s3z", | |
| "schema": "3", | |
| "root": SESSION_INPUTS / "monaco_dev/current/scaffolds_dir", | |
| "quality": SESSION_INPUTS / "monaco_dev/current/schema_quality_report/report", | |
| }, | |
| { | |
| "id": "phase1_full", | |
| "label": "Schema-v3 Phase 1 full", | |
| "short_label": "Phase 1", | |
| "job": "schema-v3-phase1-full/monaco_phase1_50doc", | |
| "schema": "3", | |
| "root": SCHEMA_REPAIR | |
| / "outputs/schema-v3-phase1-full/monaco_phase1_50doc/" | |
| "named-outputs/scaffolds_dir", | |
| "quality": SCHEMA_REPAIR | |
| / "outputs/schema-v3-phase1-full/monaco_phase1_50doc/" | |
| "named-outputs/schema_quality_report/report", | |
| }, | |
| ], | |
| }, | |
| "wiki_opentable_dev": { | |
| "label": "Open-WikiTable-dev", | |
| "source": { | |
| "kind": "raw_dir", | |
| "path": INFO | |
| / "outputs/rawtext_corpus_wikiotdev/wiki_opentable_dev/scaffolds", | |
| }, | |
| "runs": [ | |
| { | |
| "id": "legacy_full", | |
| "label": "Legacy full extraction - 50-doc subset", | |
| "short_label": "Legacy", | |
| "job": "cyan_muscle_ylws4v86bz", | |
| "schema": "legacy five-shape", | |
| "root": INFO | |
| / "outputs/e2e_dev_scaffolds/wiki_opentable_dev/named-outputs/scaffolds_dir", | |
| }, | |
| { | |
| "id": "current", | |
| "label": "Current schema-v3 extraction", | |
| "short_label": "Current", | |
| "job": "purple_eye_8f31ymzz71", | |
| "schema": "3", | |
| "root": SESSION_INPUTS / "wiki_opentable_dev/current/scaffolds_dir", | |
| "quality": SESSION_INPUTS | |
| / "wiki_opentable_dev/current/schema_quality_report/report", | |
| }, | |
| ], | |
| }, | |
| "phantom_wiki": { | |
| "label": "PhantomWiki", | |
| "source": { | |
| "kind": "corpus", | |
| "path": DATA_ROOT / "eval/phantom_wiki/unified/corpus.unified.jsonl", | |
| }, | |
| "runs": [ | |
| { | |
| "id": "legacy_full", | |
| "label": "Legacy full extraction - 50-doc subset", | |
| "short_label": "Legacy", | |
| "job": "teal_jicama_qxvpyf3jsd", | |
| "schema": "legacy five-shape", | |
| "root": INFO | |
| / "outputs/e2e_runs/new-datasets-full-20260711/phantom_wiki/" | |
| "named-outputs/scaffolds_dir", | |
| }, | |
| { | |
| "id": "current", | |
| "label": "Current schema-v3 extraction", | |
| "short_label": "Current", | |
| "job": "mango_avocado_7qnv0vlf46", | |
| "schema": "3", | |
| "root": SESSION_INPUTS / "phantom_wiki/current/scaffolds_dir", | |
| "quality": SESSION_INPUTS | |
| / "phantom_wiki/current/schema_quality_report/report", | |
| }, | |
| ], | |
| }, | |
| } | |
| def read_json(path: Path) -> Any: | |
| return json.loads(path.read_text(encoding="utf-8")) | |
| def read_jsonl(path: Path) -> Iterable[dict[str, Any]]: | |
| with path.open(encoding="utf-8") as handle: | |
| for line_number, line in enumerate(handle, 1): | |
| if not line.strip(): | |
| continue | |
| value = json.loads(line) | |
| if not isinstance(value, dict): | |
| raise ValueError(f"{path}:{line_number}: expected an object") | |
| yield value | |
| def parse_doc_id(text: str, fallback: str) -> str: | |
| for raw in text.splitlines(): | |
| line = raw.strip() | |
| if not line: | |
| continue | |
| if line.startswith("#") and line.lstrip("#").strip().lower().startswith("id:"): | |
| value = line.lstrip("#").strip().split(":", 1)[1].strip() | |
| return value or fallback | |
| return fallback | |
| return fallback | |
| def load_selected_sources(source: dict[str, Any]) -> list[dict[str, str]]: | |
| path = Path(source["path"]) | |
| if source["kind"] == "raw_dir": | |
| selected: list[Path] = [] | |
| for candidate in path.rglob("*.txt"): | |
| if candidate.is_file(): | |
| selected.append(candidate) | |
| if len(selected) == 50: | |
| break | |
| selected.sort() | |
| rows = [] | |
| for candidate in selected: | |
| text = candidate.read_text(encoding="utf-8") | |
| rows.append( | |
| { | |
| "doc_id": parse_doc_id(text, candidate.stem), | |
| "contents": text, | |
| "source_path": str(candidate.relative_to(path)), | |
| } | |
| ) | |
| return rows | |
| rows = [] | |
| for row in read_jsonl(path): | |
| rows.append( | |
| { | |
| "doc_id": str(row["id"]), | |
| "contents": str(row.get("contents", "")), | |
| "source_path": str(row["id"]), | |
| } | |
| ) | |
| if len(rows) == 50: | |
| break | |
| return rows | |
| def artifact_format(filename: str) -> str: | |
| if filename.endswith(".edges.jsonl"): | |
| return "jsonl" | |
| if filename.endswith(".timeline.json"): | |
| return "json" | |
| suffix = Path(filename).suffix.lower().lstrip(".") | |
| return suffix if suffix in {"csv", "json", "jsonl", "md"} else "text" | |
| def load_quality(path: Path | None) -> dict[str, Any] | None: | |
| if path is None or not path.exists(): | |
| return None | |
| return read_json(path) | |
| def load_scaffold_run( | |
| config: dict[str, Any], selected_doc_ids: set[str] | |
| ) -> dict[str, Any]: | |
| if config.get("status") == "pending": | |
| return { | |
| "id": config["id"], | |
| "label": config["label"], | |
| "short_label": config.get("short_label", config["label"]), | |
| "job": config["job"], | |
| "schema": config["schema"], | |
| "status": "pending", | |
| "documents": {}, | |
| "quality": None, | |
| } | |
| root = Path(config["root"]) | |
| top = read_json(root / "_index.json") | |
| documents: dict[str, list[dict[str, Any]]] = {} | |
| shape_summaries = [] | |
| for shape in top.get("shapes", []): | |
| raw_shape = str(shape["shape_id"]) | |
| canonical_shape = SHAPE_ALIASES.get(raw_shape, raw_shape) | |
| folder = str(shape.get("folder") or raw_shape).rstrip("/") | |
| shape_index = read_json(root / folder / "_index.json") | |
| shape_summaries.append( | |
| { | |
| "id": canonical_shape, | |
| "raw_id": raw_shape, | |
| "label": SHAPE_LABELS.get(canonical_shape, shape.get("display_name", raw_shape)), | |
| "description": shape_index.get("description", shape.get("definition", "")), | |
| "n_files": len(shape_index.get("entries", [])), | |
| } | |
| ) | |
| for entry in shape_index.get("entries", []): | |
| doc_id = str(entry["doc_id"]) | |
| if doc_id not in selected_doc_ids: | |
| continue | |
| filename = str(entry["file"]) | |
| artifact_path = root / folder / filename | |
| documents.setdefault(doc_id, []).append( | |
| { | |
| "shape": canonical_shape, | |
| "shape_label": SHAPE_LABELS.get(canonical_shape, canonical_shape), | |
| "raw_shape": raw_shape, | |
| "filename": filename, | |
| "format": artifact_format(filename), | |
| "unit_name": entry.get("unit_name"), | |
| "unit_description": entry.get("unit_description"), | |
| "content": artifact_path.read_text(encoding="utf-8"), | |
| } | |
| ) | |
| return { | |
| "id": config["id"], | |
| "label": config["label"], | |
| "short_label": config.get("short_label", config["label"]), | |
| "job": config["job"], | |
| "schema": config["schema"], | |
| "status": "ready", | |
| "documents": documents, | |
| "quality": load_quality(config.get("quality")), | |
| "summary": { | |
| "n_docs_seen": top.get("n_docs_seen"), | |
| "n_files_written": top.get("n_files_written"), | |
| "shapes_hash": top.get("shapes_hash"), | |
| "shape_summaries": shape_summaries, | |
| "validation_metrics": top.get("validation_metrics"), | |
| }, | |
| } | |
| def record_name(doc_id: str) -> str: | |
| return hashlib.sha1(doc_id.encode()).hexdigest() + ".json" | |
| def build_dataset(dataset_id: str, config: dict[str, Any], data_dir: Path) -> dict[str, Any]: | |
| sources = load_selected_sources(config["source"]) | |
| if len(sources) != 50: | |
| raise ValueError(f"{dataset_id}: expected 50 source documents, got {len(sources)}") | |
| source_ids = [row["doc_id"] for row in sources] | |
| if len(set(source_ids)) != 50: | |
| raise ValueError(f"{dataset_id}: selected source IDs are not unique") | |
| runs = [load_scaffold_run(run, set(source_ids)) for run in config["runs"]] | |
| records_dir = data_dir / "records" / dataset_id | |
| records_dir.mkdir(parents=True, exist_ok=True) | |
| index_rows = [] | |
| for position, source in enumerate(sources, 1): | |
| doc_id = source["doc_id"] | |
| run_records = {} | |
| for run in runs: | |
| artifacts = run["documents"].get(doc_id, []) | |
| run_records[run["id"]] = { | |
| "status": run["status"], | |
| "artifacts": artifacts, | |
| "n_artifacts": len(artifacts), | |
| } | |
| digest = hashlib.sha256(source["contents"].encode()).hexdigest() | |
| record = { | |
| "dataset": dataset_id, | |
| "dataset_label": config["label"], | |
| "position": position, | |
| "doc_id": doc_id, | |
| "source": { | |
| "contents": source["contents"], | |
| "source_path": source["source_path"], | |
| "sha256": digest, | |
| }, | |
| "runs": run_records, | |
| } | |
| filename = record_name(doc_id) | |
| (records_dir / filename).write_text( | |
| json.dumps(record, ensure_ascii=False), encoding="utf-8" | |
| ) | |
| index_rows.append( | |
| { | |
| "position": position, | |
| "doc_id": doc_id, | |
| "record": f"data/records/{dataset_id}/{filename}", | |
| "source_preview": source["contents"].replace("\n", " ")[:180], | |
| "source_sha256": digest, | |
| "run_counts": { | |
| run_id: value["n_artifacts"] for run_id, value in run_records.items() | |
| }, | |
| } | |
| ) | |
| return { | |
| "id": dataset_id, | |
| "label": config["label"], | |
| "n_docs": len(sources), | |
| "runs": [ | |
| { | |
| key: run.get(key) | |
| for key in ( | |
| "id", | |
| "label", | |
| "short_label", | |
| "job", | |
| "schema", | |
| "status", | |
| "quality", | |
| "summary", | |
| ) | |
| } | |
| for run in runs | |
| ], | |
| "records": index_rows, | |
| } | |
| def validate_bundle(index: dict[str, Any], data_dir: Path) -> None: | |
| if len(index["datasets"]) != 3: | |
| raise ValueError("expected three datasets") | |
| for dataset in index["datasets"]: | |
| if dataset["n_docs"] != 50 or len(dataset["records"]) != 50: | |
| raise ValueError(f"{dataset['id']}: expected 50 records") | |
| if len({row["doc_id"] for row in dataset["records"]}) != 50: | |
| raise ValueError(f"{dataset['id']}: duplicate document IDs") | |
| for row in dataset["records"]: | |
| path = ROOT / row["record"] | |
| if not path.exists(): | |
| raise ValueError(f"missing record shard: {path}") | |
| phantom = next(row for row in index["datasets"] if row["id"] == "phantom_wiki") | |
| current = next(run for run in phantom["runs"] if run["id"] == "current") | |
| if current["status"] != "ready": | |
| raise ValueError("PhantomWiki current run must be ready") | |
| def main() -> int: | |
| parser = argparse.ArgumentParser(description=__doc__) | |
| parser.add_argument("--out", type=Path, default=ROOT / "data") | |
| args = parser.parse_args() | |
| if args.out.exists(): | |
| shutil.rmtree(args.out) | |
| args.out.mkdir(parents=True) | |
| index = { | |
| "title": "Structure Generation Prompt Viewer", | |
| "datasets": [ | |
| build_dataset(dataset_id, config, args.out) | |
| for dataset_id, config in DATASETS.items() | |
| ], | |
| } | |
| (args.out / "index.json").write_text( | |
| json.dumps(index, ensure_ascii=False, indent=2) + "\n", encoding="utf-8" | |
| ) | |
| validate_bundle(index, args.out) | |
| print( | |
| "Built 150 document records: " | |
| + ", ".join(f"{row['label']}={row['n_docs']}" for row in index["datasets"]) | |
| ) | |
| return 0 | |
| if __name__ == "__main__": | |
| raise SystemExit(main()) | |