File size: 11,050 Bytes
3d968d9
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
"""Replay scenario.json against the real app.py and print the results as JSON.

This is the reference half of the parity check. It imports the vendored copy of
upstream's server from ../../app, points its storage at a temporary directory,
and replaces its two sources of nondeterminism -- the clock and uuid4 -- with
counters so that a run is byte-for-byte reproducible.

Making the clock and the ids sequential does more than remove noise: it means
the port has to call them the same number of times, in the same order. A missing
utc_now() or an extra uuid4() in the JavaScript shows up as a mismatch instead
of hiding behind a value nobody compares.

Usage (normally driven by run-parity.mjs):

    python tests/parity/reference.py            # results to stdout
    python tests/parity/reference.py --out x.json

Exit status is 0 when the scenario ran, regardless of the individual response
statuses -- rejections are part of what is being compared.
"""

from __future__ import annotations

import argparse
import hashlib
import io
import itertools
import json
import shutil
import sys
import tempfile
import threading
import urllib.error
import urllib.request
import zipfile
from datetime import datetime, timedelta, timezone
from http.server import HTTPServer
from pathlib import Path

HERE = Path(__file__).resolve().parent
REPO_ROOT = HERE.parent.parent
APP_DIR = REPO_ROOT / "app"
GENERATED_DIR = HERE / ".corpus"

# The instant the fake clock starts from. Any fixed value works; this one is
# obviously synthetic so it cannot be mistaken for a real timestamp.
CLOCK_START = datetime(2026, 1, 2, 3, 4, 5, tzinfo=timezone.utc)

BOUNDARY = "----SimpleDocReviewParityBoundary"


# --------------------------------------------------------------- fixtures

def make_docx(paragraphs: list[str], compression: int) -> bytes:
    """A minimal DOCX: one word/document.xml, stored or deflated."""
    body = "".join(
        f"<w:p><w:r><w:t>{paragraph}</w:t></w:r></w:p>" for paragraph in paragraphs
    )
    document_xml = (
        '<?xml version="1.0" encoding="UTF-8" standalone="yes"?>'
        '<w:document xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main">'
        f"<w:body>{body}</w:body>"
        "</w:document>"
    )
    buffer = io.BytesIO()
    with zipfile.ZipFile(buffer, "w", compression) as archive:
        # A real DOCX carries more parts than this; extract_docx_text reads only
        # word/document.xml, so the extra parts would test nothing. One is added
        # anyway so the archive has more than a single entry to walk past.
        archive.writestr("[Content_Types].xml", "<Types/>")
        archive.writestr("word/document.xml", document_xml)
    return buffer.getvalue()


def write_generated_fixtures() -> None:
    """Create the binary and encoding-specific fixtures scenario.json expects."""
    GENERATED_DIR.mkdir(exist_ok=True)

    paragraphs = [
        "Purpose: confirm the DOCX text extractor agrees with the browser port.",
        "Scope: paragraphs, entity escapes such as R&amp;D and &lt;brackets&gt;, and blank runs.",
        "",
        "Owner: Document Control",
        "Date: 2026-04-01",
        "Risk: a divergent extractor would change every finding downstream.",
        "Approval: pending sign-off.",
    ]
    (GENERATED_DIR / "sample-deflate.docx").write_bytes(
        make_docx(paragraphs, zipfile.ZIP_DEFLATED)
    )
    (GENERATED_DIR / "sample-stored.docx").write_bytes(
        make_docx(paragraphs, zipfile.ZIP_STORED)
    )

    # A ZIP with no word/document.xml: "does not contain document text".
    buffer = io.BytesIO()
    with zipfile.ZipFile(buffer, "w") as archive:
        archive.writestr("docProps/core.xml", "<coreProperties/>")
    (GENERATED_DIR / "no-document-xml.docx").write_bytes(buffer.getvalue())

    # Not a ZIP at all: BadZipFile -> "appears to be invalid".
    (GENERATED_DIR / "broken.docx").write_bytes(b"PK\x03\x04 truncated nonsense")

    # UTF-16 with a BOM, and cp1252 with bytes that are not valid UTF-8, so the
    # decoder fallback chain in decode_text_bytes gets exercised.
    (GENERATED_DIR / "utf16.txt").write_bytes(
        (
            "Purpose: check the UTF-16 decoder.\n"
            "Owner: Encoding Team\n"
            "Approval: signature on file.\n"
        ).encode("utf-16")
    )
    # Curly quotes and an en dash become 0x93, 0x94, 0x96 in cp1252 - byte
    # values that are not valid UTF-8, so decode_text_bytes has to fall through
    # to its third attempt.
    (GENERATED_DIR / "cp1252.txt").write_bytes(
        (
            "Purpose: check the cp1252 fallback.\n"
            "Scope: “smart quotes” and an en dash – here.\n"
            "Owner: Encoding Team\n"
            "Approval: signature on file.\n"
        ).encode("cp1252")
    )

    (GENERATED_DIR / "empty.txt").write_bytes(b"")


# ------------------------------------------------- deterministic app.py

def load_app(storage_root: Path):
    """Import upstream's server with its clock, ids, and storage replaced."""
    sys.path.insert(0, str(APP_DIR))
    import app  # noqa: E402  (the path has to be set up first)

    clock = itertools.count()
    ids = itertools.count(1)

    def utc_now() -> str:
        return (CLOCK_START + timedelta(seconds=next(clock))).isoformat(timespec="seconds")

    class FakeUuid:
        def __init__(self, value: int) -> None:
            self.hex = f"{value:032x}"

    class FakeUuidModule:
        @staticmethod
        def uuid4() -> FakeUuid:
            return FakeUuid(next(ids))

    app.utc_now = utc_now
    app.uuid = FakeUuidModule()

    app.DATA_DIR = storage_root / "data"
    app.UPLOAD_DIR = storage_root / "uploads"
    app.DB_PATH = app.DATA_DIR / "reviews.json"

    # Keep the request log out of stdout, which carries the JSON results.
    app.ReviewHandler.log_message = lambda self, fmt, *args: None
    return app


def start_server(app_module):
    """A single-threaded server, so requests cannot interleave the counters."""
    server = HTTPServer(("127.0.0.1", 0), app_module.ReviewHandler)
    thread = threading.Thread(target=server.serve_forever, daemon=True)
    thread.start()
    return server, f"http://127.0.0.1:{server.server_address[1]}"


# ------------------------------------------------------------ scenario

def resolve(value, results):
    """Substitute ${stepName|dotted.path} against earlier response bodies."""
    if isinstance(value, str):
        while True:
            start = value.find("${")
            if start < 0:
                return value
            end = value.index("}", start)
            token = value[start + 2:end]
            step, _, path = token.partition("|")
            current = results[step]["json"]
            for part in [p for p in path.split(".") if p]:
                current = current[int(part)] if part.isdigit() else current[part]
            value = value[:start] + str(current) + value[end + 1:]
    if isinstance(value, list):
        return [resolve(item, results) for item in value]
    if isinstance(value, dict):
        return {key: resolve(item, results) for key, item in value.items()}
    return value


def multipart_body(filename: str, content: bytes, fields: dict) -> tuple[bytes, str]:
    parts: list[bytes] = []
    parts.append(
        f'--{BOUNDARY}\r\nContent-Disposition: form-data; name="file"; '
        f'filename="{filename}"\r\nContent-Type: application/octet-stream\r\n\r\n'.encode("utf-8")
        + content
        + b"\r\n"
    )
    for key, value in fields.items():
        parts.append(
            f'--{BOUNDARY}\r\nContent-Disposition: form-data; name="{key}"\r\n\r\n'.encode("utf-8")
            + str(value).encode("utf-8")
            + b"\r\n"
        )
    parts.append(f"--{BOUNDARY}--\r\n".encode("utf-8"))
    return b"".join(parts), f"multipart/form-data; boundary={BOUNDARY}"


def request(base_url: str, step: dict, results: dict) -> dict:
    path = resolve(step["path"], results)
    method = step.get("method", "GET")
    data = None
    headers = {}

    if "upload" in step:
        upload = step["upload"]
        content = (HERE / upload["file"]).read_bytes()
        fields = resolve(upload.get("fields", {}), results)
        data, content_type = multipart_body(upload["filename"], content, fields)
        headers["Content-Type"] = content_type
    elif "body" in step:
        data = json.dumps(resolve(step["body"], results)).encode("utf-8")
        headers["Content-Type"] = "application/json"

    http_request = urllib.request.Request(base_url + path, data=data, method=method, headers=headers)
    try:
        with urllib.request.urlopen(http_request) as response:
            status = response.status
            payload = response.read()
            response_headers = dict(response.headers.items())
    except urllib.error.HTTPError as error:
        status = error.code
        payload = error.read()
        response_headers = dict(error.headers.items())

    record: dict = {
        "name": step["name"],
        "status": status,
        "contentType": response_headers.get("Content-Type"),
    }
    if step.get("binary") and status == 200:
        record["sha256"] = hashlib.sha256(payload).hexdigest()
        record["bytes"] = len(payload)
        record["contentDisposition"] = response_headers.get("Content-Disposition")
        record["json"] = None
    else:
        record["json"] = json.loads(payload.decode("utf-8")) if payload else None
    return record


def main() -> int:
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument("--out", help="write the results here instead of stdout")
    args = parser.parse_args()

    write_generated_fixtures()

    scenario = json.loads((HERE / "scenario.json").read_text(encoding="utf-8"))
    storage_root = Path(tempfile.mkdtemp(prefix="sdr-parity-"))

    try:
        app_module = load_app(storage_root)
        server, base_url = start_server(app_module)
        try:
            results: dict = {}
            ordered: list[dict] = []
            for step in scenario["steps"]:
                record = request(base_url, step, results)
                results[step["name"]] = record
                ordered.append(record)
        finally:
            server.shutdown()
            server.server_close()

        # The final store, so fields no response exposes (stored_name, the
        # extracted text, per-assignment comment history) are compared too.
        final_db = json.loads(app_module.DB_PATH.read_text(encoding="utf-8"))
        uploads = sorted(path.name for path in app_module.UPLOAD_DIR.iterdir())

        payload = {"steps": ordered, "db": final_db, "uploads": uploads}
        text = json.dumps(payload, indent=2, sort_keys=True, ensure_ascii=False)
        if args.out:
            Path(args.out).write_text(text, encoding="utf-8")
        else:
            sys.stdout.write(text)
        return 0
    finally:
        shutil.rmtree(storage_root, ignore_errors=True)


if __name__ == "__main__":
    raise SystemExit(main())