Spaces:
Running
Running
Eric Putnam
Claude Opus 5 (1M context)
feat: deploy simpledocreview as a browser-only static space
3d968d9 Download tests/parity/reference.py from Hoctar77/SimpleDocReview: direct link, hf CLI and curl.
- Browser
- Download file 11.1 kB
-
https://huggingface.co/spaces/Hoctar77/SimpleDocReview/resolve/main/tests/parity/reference.py
- Command line
-
hf download hf://spaces/Hoctar77/SimpleDocReview/tests/parity/reference.py
-
curl -L -o reference.py https://huggingface.co/spaces/Hoctar77/SimpleDocReview/resolve/main/tests/parity/reference.py
11.1 kB
| """Replay scenario.json against the real app.py and print the results as JSON. | |
| This is the reference half of the parity check. It imports the vendored copy of | |
| upstream's server from ../../app, points its storage at a temporary directory, | |
| and replaces its two sources of nondeterminism -- the clock and uuid4 -- with | |
| counters so that a run is byte-for-byte reproducible. | |
| Making the clock and the ids sequential does more than remove noise: it means | |
| the port has to call them the same number of times, in the same order. A missing | |
| utc_now() or an extra uuid4() in the JavaScript shows up as a mismatch instead | |
| of hiding behind a value nobody compares. | |
| Usage (normally driven by run-parity.mjs): | |
| python tests/parity/reference.py # results to stdout | |
| python tests/parity/reference.py --out x.json | |
| Exit status is 0 when the scenario ran, regardless of the individual response | |
| statuses -- rejections are part of what is being compared. | |
| """ | |
| from __future__ import annotations | |
| import argparse | |
| import hashlib | |
| import io | |
| import itertools | |
| import json | |
| import shutil | |
| import sys | |
| import tempfile | |
| import threading | |
| import urllib.error | |
| import urllib.request | |
| import zipfile | |
| from datetime import datetime, timedelta, timezone | |
| from http.server import HTTPServer | |
| from pathlib import Path | |
| HERE = Path(__file__).resolve().parent | |
| REPO_ROOT = HERE.parent.parent | |
| APP_DIR = REPO_ROOT / "app" | |
| GENERATED_DIR = HERE / ".corpus" | |
| # The instant the fake clock starts from. Any fixed value works; this one is | |
| # obviously synthetic so it cannot be mistaken for a real timestamp. | |
| CLOCK_START = datetime(2026, 1, 2, 3, 4, 5, tzinfo=timezone.utc) | |
| BOUNDARY = "----SimpleDocReviewParityBoundary" | |
| # --------------------------------------------------------------- fixtures | |
| def make_docx(paragraphs: list[str], compression: int) -> bytes: | |
| """A minimal DOCX: one word/document.xml, stored or deflated.""" | |
| body = "".join( | |
| f"<w:p><w:r><w:t>{paragraph}</w:t></w:r></w:p>" for paragraph in paragraphs | |
| ) | |
| document_xml = ( | |
| '<?xml version="1.0" encoding="UTF-8" standalone="yes"?>' | |
| '<w:document xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main">' | |
| f"<w:body>{body}</w:body>" | |
| "</w:document>" | |
| ) | |
| buffer = io.BytesIO() | |
| with zipfile.ZipFile(buffer, "w", compression) as archive: | |
| # A real DOCX carries more parts than this; extract_docx_text reads only | |
| # word/document.xml, so the extra parts would test nothing. One is added | |
| # anyway so the archive has more than a single entry to walk past. | |
| archive.writestr("[Content_Types].xml", "<Types/>") | |
| archive.writestr("word/document.xml", document_xml) | |
| return buffer.getvalue() | |
| def write_generated_fixtures() -> None: | |
| """Create the binary and encoding-specific fixtures scenario.json expects.""" | |
| GENERATED_DIR.mkdir(exist_ok=True) | |
| paragraphs = [ | |
| "Purpose: confirm the DOCX text extractor agrees with the browser port.", | |
| "Scope: paragraphs, entity escapes such as R&D and <brackets>, and blank runs.", | |
| "", | |
| "Owner: Document Control", | |
| "Date: 2026-04-01", | |
| "Risk: a divergent extractor would change every finding downstream.", | |
| "Approval: pending sign-off.", | |
| ] | |
| (GENERATED_DIR / "sample-deflate.docx").write_bytes( | |
| make_docx(paragraphs, zipfile.ZIP_DEFLATED) | |
| ) | |
| (GENERATED_DIR / "sample-stored.docx").write_bytes( | |
| make_docx(paragraphs, zipfile.ZIP_STORED) | |
| ) | |
| # A ZIP with no word/document.xml: "does not contain document text". | |
| buffer = io.BytesIO() | |
| with zipfile.ZipFile(buffer, "w") as archive: | |
| archive.writestr("docProps/core.xml", "<coreProperties/>") | |
| (GENERATED_DIR / "no-document-xml.docx").write_bytes(buffer.getvalue()) | |
| # Not a ZIP at all: BadZipFile -> "appears to be invalid". | |
| (GENERATED_DIR / "broken.docx").write_bytes(b"PK\x03\x04 truncated nonsense") | |
| # UTF-16 with a BOM, and cp1252 with bytes that are not valid UTF-8, so the | |
| # decoder fallback chain in decode_text_bytes gets exercised. | |
| (GENERATED_DIR / "utf16.txt").write_bytes( | |
| ( | |
| "Purpose: check the UTF-16 decoder.\n" | |
| "Owner: Encoding Team\n" | |
| "Approval: signature on file.\n" | |
| ).encode("utf-16") | |
| ) | |
| # Curly quotes and an en dash become 0x93, 0x94, 0x96 in cp1252 - byte | |
| # values that are not valid UTF-8, so decode_text_bytes has to fall through | |
| # to its third attempt. | |
| (GENERATED_DIR / "cp1252.txt").write_bytes( | |
| ( | |
| "Purpose: check the cp1252 fallback.\n" | |
| "Scope: “smart quotes” and an en dash – here.\n" | |
| "Owner: Encoding Team\n" | |
| "Approval: signature on file.\n" | |
| ).encode("cp1252") | |
| ) | |
| (GENERATED_DIR / "empty.txt").write_bytes(b"") | |
| # ------------------------------------------------- deterministic app.py | |
| def load_app(storage_root: Path): | |
| """Import upstream's server with its clock, ids, and storage replaced.""" | |
| sys.path.insert(0, str(APP_DIR)) | |
| import app # noqa: E402 (the path has to be set up first) | |
| clock = itertools.count() | |
| ids = itertools.count(1) | |
| def utc_now() -> str: | |
| return (CLOCK_START + timedelta(seconds=next(clock))).isoformat(timespec="seconds") | |
| class FakeUuid: | |
| def __init__(self, value: int) -> None: | |
| self.hex = f"{value:032x}" | |
| class FakeUuidModule: | |
| def uuid4() -> FakeUuid: | |
| return FakeUuid(next(ids)) | |
| app.utc_now = utc_now | |
| app.uuid = FakeUuidModule() | |
| app.DATA_DIR = storage_root / "data" | |
| app.UPLOAD_DIR = storage_root / "uploads" | |
| app.DB_PATH = app.DATA_DIR / "reviews.json" | |
| # Keep the request log out of stdout, which carries the JSON results. | |
| app.ReviewHandler.log_message = lambda self, fmt, *args: None | |
| return app | |
| def start_server(app_module): | |
| """A single-threaded server, so requests cannot interleave the counters.""" | |
| server = HTTPServer(("127.0.0.1", 0), app_module.ReviewHandler) | |
| thread = threading.Thread(target=server.serve_forever, daemon=True) | |
| thread.start() | |
| return server, f"http://127.0.0.1:{server.server_address[1]}" | |
| # ------------------------------------------------------------ scenario | |
| def resolve(value, results): | |
| """Substitute ${stepName|dotted.path} against earlier response bodies.""" | |
| if isinstance(value, str): | |
| while True: | |
| start = value.find("${") | |
| if start < 0: | |
| return value | |
| end = value.index("}", start) | |
| token = value[start + 2:end] | |
| step, _, path = token.partition("|") | |
| current = results[step]["json"] | |
| for part in [p for p in path.split(".") if p]: | |
| current = current[int(part)] if part.isdigit() else current[part] | |
| value = value[:start] + str(current) + value[end + 1:] | |
| if isinstance(value, list): | |
| return [resolve(item, results) for item in value] | |
| if isinstance(value, dict): | |
| return {key: resolve(item, results) for key, item in value.items()} | |
| return value | |
| def multipart_body(filename: str, content: bytes, fields: dict) -> tuple[bytes, str]: | |
| parts: list[bytes] = [] | |
| parts.append( | |
| f'--{BOUNDARY}\r\nContent-Disposition: form-data; name="file"; ' | |
| f'filename="{filename}"\r\nContent-Type: application/octet-stream\r\n\r\n'.encode("utf-8") | |
| + content | |
| + b"\r\n" | |
| ) | |
| for key, value in fields.items(): | |
| parts.append( | |
| f'--{BOUNDARY}\r\nContent-Disposition: form-data; name="{key}"\r\n\r\n'.encode("utf-8") | |
| + str(value).encode("utf-8") | |
| + b"\r\n" | |
| ) | |
| parts.append(f"--{BOUNDARY}--\r\n".encode("utf-8")) | |
| return b"".join(parts), f"multipart/form-data; boundary={BOUNDARY}" | |
| def request(base_url: str, step: dict, results: dict) -> dict: | |
| path = resolve(step["path"], results) | |
| method = step.get("method", "GET") | |
| data = None | |
| headers = {} | |
| if "upload" in step: | |
| upload = step["upload"] | |
| content = (HERE / upload["file"]).read_bytes() | |
| fields = resolve(upload.get("fields", {}), results) | |
| data, content_type = multipart_body(upload["filename"], content, fields) | |
| headers["Content-Type"] = content_type | |
| elif "body" in step: | |
| data = json.dumps(resolve(step["body"], results)).encode("utf-8") | |
| headers["Content-Type"] = "application/json" | |
| http_request = urllib.request.Request(base_url + path, data=data, method=method, headers=headers) | |
| try: | |
| with urllib.request.urlopen(http_request) as response: | |
| status = response.status | |
| payload = response.read() | |
| response_headers = dict(response.headers.items()) | |
| except urllib.error.HTTPError as error: | |
| status = error.code | |
| payload = error.read() | |
| response_headers = dict(error.headers.items()) | |
| record: dict = { | |
| "name": step["name"], | |
| "status": status, | |
| "contentType": response_headers.get("Content-Type"), | |
| } | |
| if step.get("binary") and status == 200: | |
| record["sha256"] = hashlib.sha256(payload).hexdigest() | |
| record["bytes"] = len(payload) | |
| record["contentDisposition"] = response_headers.get("Content-Disposition") | |
| record["json"] = None | |
| else: | |
| record["json"] = json.loads(payload.decode("utf-8")) if payload else None | |
| return record | |
| def main() -> int: | |
| parser = argparse.ArgumentParser(description=__doc__) | |
| parser.add_argument("--out", help="write the results here instead of stdout") | |
| args = parser.parse_args() | |
| write_generated_fixtures() | |
| scenario = json.loads((HERE / "scenario.json").read_text(encoding="utf-8")) | |
| storage_root = Path(tempfile.mkdtemp(prefix="sdr-parity-")) | |
| try: | |
| app_module = load_app(storage_root) | |
| server, base_url = start_server(app_module) | |
| try: | |
| results: dict = {} | |
| ordered: list[dict] = [] | |
| for step in scenario["steps"]: | |
| record = request(base_url, step, results) | |
| results[step["name"]] = record | |
| ordered.append(record) | |
| finally: | |
| server.shutdown() | |
| server.server_close() | |
| # The final store, so fields no response exposes (stored_name, the | |
| # extracted text, per-assignment comment history) are compared too. | |
| final_db = json.loads(app_module.DB_PATH.read_text(encoding="utf-8")) | |
| uploads = sorted(path.name for path in app_module.UPLOAD_DIR.iterdir()) | |
| payload = {"steps": ordered, "db": final_db, "uploads": uploads} | |
| text = json.dumps(payload, indent=2, sort_keys=True, ensure_ascii=False) | |
| if args.out: | |
| Path(args.out).write_text(text, encoding="utf-8") | |
| else: | |
| sys.stdout.write(text) | |
| return 0 | |
| finally: | |
| shutil.rmtree(storage_root, ignore_errors=True) | |
| if __name__ == "__main__": | |
| raise SystemExit(main()) | |