File size: 14,743 Bytes
58c2da3
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
from __future__ import annotations

import hashlib
import json
from datetime import UTC, datetime
from pathlib import Path

import httpx
import pytest

from gcmd_classifier.config import DatasetDocumentSettings
from gcmd_classifier.datasets.cmr import resolve_cmr_collection, validate_cmr_url
from gcmd_classifier.datasets.documents import retrieve_selected_readme, validate_pdf_bytes
from gcmd_classifier.datasets.errors import (
    READMEDocumentError,
    READMERetrievalError,
    READMESelectionError,
)
from gcmd_classifier.datasets.models import BlindCollectionView, CMRRetrievedSource, DatasetIdentity
from gcmd_classifier.datasets.readme_discovery import (
    discover_readme_candidates,
    select_readme_candidate,
)
from tests.datasets.fixture_manifest import load_fixture_manifest

URL = "https://docs.example.test/readme.pdf"
NOW = datetime(2026, 8, 13, 13, 0, tzinfo=UTC)
PDF = b"%PDF-1.7\n1 0 obj\n<<>>\nendobj\nstartxref\n9\n%%EOF\n"
PUBLIC4 = "93.184.216.34"
PUBLIC6 = "2606:2800:220:1:248:1893:25c8:1946"


def _selection(url: str = URL):
    view = BlindCollectionView(
        identity=DatasetIdentity(
            concept_id="C1-P",
            native_id="TARGET_001",
            short_name="TARGET",
            version="001",
            cmr_revision_id=4,
        ),
        derived_native_id="TARGET_001",
        related_urls=({"Subtype": "READ-ME", "URL": url, "Description": "README"},),
        blind_view_sha256="a" * 64,
    )
    discovery = discover_readme_candidates(view, cmr_source_sha256="b" * 64)
    return select_readme_candidate(discovery, [(0, url)], utc_now=lambda: NOW)


def _client(handler) -> httpx.Client:
    return httpx.Client(transport=httpx.MockTransport(handler), follow_redirects=False)


def _ok(
    request: httpx.Request,
    *,
    body: bytes = PDF,
    content_type: str = "application/pdf",
    headers=None,
):
    return httpx.Response(
        200,
        headers={"content-type": content_type, **(headers or {})},
        content=body,
        request=request,
    )


def _retrieve(
    tmp_path: Path, handler=_ok, *, resolver=lambda host: [PUBLIC4], settings=None, selection=None
):
    with _client(handler) as client:
        return retrieve_selected_readme(
            selection or _selection(),
            client=client,
            resolver=resolver,
            artifact_directory=tmp_path,
            settings=settings,
            utc_now=lambda: NOW,
        )


@pytest.mark.parametrize("addresses", ([PUBLIC4], [PUBLIC6], [PUBLIC4, PUBLIC6]))
def test_public_ipv4_ipv6_and_mixed_public_addresses_are_allowed(
    tmp_path: Path, addresses: list[str]
) -> None:
    result = _retrieve(tmp_path, resolver=lambda host: addresses)
    assert result.content == PDF
    assert result.record.artifact.sha256 == hashlib.sha256(PDF).hexdigest()
    storage_reference = result.record.artifact.storage_reference
    assert storage_reference is not None
    assert (tmp_path / storage_reference).read_bytes() == PDF
    assert not Path(storage_reference).is_absolute()


@pytest.mark.parametrize(
    "address",
    (
        "10.0.0.1",
        "127.0.0.1",
        "169.254.1.1",
        "192.0.2.1",
        "0.0.0.0",
        "224.0.0.1",
        "169.254.169.254",
        "::1",
        "fe80::1",
        "2001:db8::1",
        "::",
    ),
)
def test_non_public_destinations_are_rejected_without_http(tmp_path: Path, address: str) -> None:
    calls = 0

    def handler(request):
        nonlocal calls
        calls += 1
        return _ok(request)

    with pytest.raises(READMERetrievalError) as captured:
        _retrieve(tmp_path, handler, resolver=lambda host: [address])
    assert captured.value.code == "README_NON_PUBLIC_ADDRESS"
    assert calls == 0


def test_mixed_public_nonpublic_and_dns_rebinding_are_rejected(tmp_path: Path) -> None:
    with pytest.raises(READMERetrievalError):
        _retrieve(tmp_path, resolver=lambda host: [PUBLIC4, "10.0.0.1"])
    answers = iter(([PUBLIC4], ["127.0.0.1"]))
    with pytest.raises(READMERetrievalError):
        _retrieve(tmp_path, resolver=lambda host: next(answers))


def test_valid_redirect_revalidates_destination_and_preserves_history(tmp_path: Path) -> None:
    target = "https://cdn.example.test/file"
    requests = []

    def handler(request):
        requests.append(str(request.url))
        return (
            httpx.Response(302, headers={"location": target}, request=request)
            if len(requests) == 1
            else _ok(request)
        )

    result = _retrieve(tmp_path, handler)
    assert requests == [URL, target]
    assert result.record.final_url == target
    assert result.record.redirects[0].target_url == target


@pytest.mark.parametrize(
    "target",
    (
        "ftp://example.test/a.pdf",
        "https://example.test:444/a.pdf",
        "https://u:p@example.test/a.pdf",
        "https://example.test/a.pdf#x",
        "https://example.test/a.pdf?token=x",
        "https://example.test/login",
    ),
)
def test_unsafe_and_authentication_redirects_are_rejected(tmp_path: Path, target: str) -> None:
    def handler(request):
        return httpx.Response(302, headers={"location": target}, request=request)

    with pytest.raises(READMERetrievalError):
        _retrieve(tmp_path, handler)


def test_redirect_loop_and_limit(tmp_path: Path) -> None:
    second = "https://docs.example.test/second.pdf"

    def loop(request):
        target = second if str(request.url) == URL else URL
        return httpx.Response(302, headers={"location": target}, request=request)

    with pytest.raises(READMERetrievalError) as captured:
        _retrieve(tmp_path, loop)
    assert captured.value.code == "README_REDIRECT_LOOP"
    with pytest.raises(READMERetrievalError) as captured:
        _retrieve(tmp_path, loop, settings=DatasetDocumentSettings(max_redirects=0))
    assert captured.value.code == "README_REDIRECT_LIMIT"


@pytest.mark.parametrize("status", (401, 403))
def test_authentication_status_is_not_public(tmp_path: Path, status: int) -> None:
    with pytest.raises(READMERetrievalError) as captured:
        _retrieve(tmp_path, lambda request: httpx.Response(status, request=request))
    assert captured.value.code == "README_NOT_PUBLIC"


def test_login_html_is_not_public(tmp_path: Path) -> None:
    with pytest.raises(READMERetrievalError) as captured:
        _retrieve(
            tmp_path,
            lambda request: _ok(
                request,
                body=b"<html><form><input type='password'></form></html>",
                content_type="text/html",
            ),
        )
    assert captured.value.code == "README_NOT_PUBLIC"


@pytest.mark.parametrize(
    "exc", (httpx.ReadTimeout("x"), httpx.ConnectError("x"), httpx.ConnectError("tls"))
)
def test_transport_failures_are_typed_and_preserve_no_false_success(
    tmp_path: Path, exc: Exception
) -> None:
    def handler(request):
        raise exc

    with pytest.raises(READMERetrievalError):
        _retrieve(tmp_path, handler)


def test_size_content_length_truncation_and_operation_limits(tmp_path: Path) -> None:
    settings = DatasetDocumentSettings(max_download_bytes=10)
    with pytest.raises(READMERetrievalError) as captured:
        _retrieve(tmp_path, settings=settings)
    assert captured.value.code == "README_TOO_LARGE"

    def truncated(request):
        return _ok(request, headers={"content-length": str(len(PDF) + 1)})

    with pytest.raises(READMERetrievalError) as captured:
        _retrieve(tmp_path, truncated)
    assert captured.value.code == "README_TRUNCATED"


@pytest.mark.parametrize(
    ("body", "media"),
    (
        (b"<html>not pdf</html>", "text/html"),
        (b"{}", "application/json"),
        (b"plain", "text/plain"),
        (b"", "application/pdf"),
        (b"%PDF-1.7 malformed", "application/pdf"),
        (b"PNG", "application/pdf"),
    ),
)
def test_unsupported_and_misleading_content_is_rejected(body: bytes, media: str) -> None:
    with pytest.raises(READMEDocumentError):
        validate_pdf_bytes(body, media)


@pytest.mark.parametrize("media", ("application/pdf", "application/octet-stream", None))
def test_pdf_byte_validation_is_extension_independent_and_stable(media: str | None) -> None:
    assert validate_pdf_bytes(PDF, media)[0] == "application/pdf"


def test_raw_url_local_path_and_tampered_selection_cannot_enter_retriever(tmp_path: Path) -> None:
    with _client(_ok) as client, pytest.raises(READMESelectionError):
        retrieve_selected_readme(
            URL, client=client, resolver=lambda host: [PUBLIC4], artifact_directory=tmp_path
        )
    selected = _selection()
    changed = selected.model_copy(update={"selected_url": "tests/fixtures/datasets/pdf/file.pdf"})
    with _client(_ok) as client, pytest.raises(READMESelectionError):
        retrieve_selected_readme(
            changed, client=client, resolver=lambda host: [PUBLIC4], artifact_directory=tmp_path
        )


def test_failure_preserves_partial_artifact_and_performs_no_cleanup(tmp_path: Path) -> None:
    with pytest.raises(READMEDocumentError):
        _retrieve(
            tmp_path,
            lambda request: _ok(request, body=b"not a pdf", content_type="application/pdf"),
        )
    partials = list(tmp_path.rglob("document.partial"))
    assert len(partials) == 1 and partials[0].read_bytes() == b"not a pdf"


def test_all_frozen_cases_require_discovery_selection_and_exact_mocked_bytes(
    tmp_path: Path,
) -> None:
    manifest, cases = load_fixture_manifest(Path("tests/fixtures/datasets/cases.json"))
    before = {
        path: path.read_bytes() for case in cases for path in (case.cmr_path, case.readme_path)
    }
    assert len(manifest.cases) == 4
    for case in cases:
        meta = case.metadata
        cmr_bytes = case.cmr_path.read_bytes()
        pdf_bytes = case.readme_path.read_bytes()
        validated = validate_cmr_url(meta.cmr_url)
        source = CMRRetrievedSource(
            submitted_url=meta.cmr_url,
            final_url=meta.cmr_url,
            retrieved_at="2026-08-13T00:00:00Z",
            status_code=200,
            response_bytes=cmr_bytes,
            source_text=cmr_bytes.decode(),
            sha256=hashlib.sha256(cmr_bytes).hexdigest(),
            parsed_json=json.loads(cmr_bytes),
        )
        resolved = resolve_cmr_collection(validated, source)
        discovery = discover_readme_candidates(resolved.blind_view, cmr_source_sha256=source.sha256)
        matching = [c for c in discovery.candidates if c.url == meta.selected_readme_url]
        assert len(matching) == 1
        selected = select_readme_candidate(
            discovery, [(matching[0].source_index, matching[0].url)], utc_now=lambda: NOW
        )

        def exact_transport(request, expected=meta.selected_readme_url, content=pdf_bytes):
            assert str(request.url) == expected
            return _ok(request, body=content)

        result = _retrieve(tmp_path / meta.case_id, exact_transport, selection=selected)
        assert result.content == pdf_bytes
        assert result.record.artifact.sha256 == meta.readme_sha256
        assert result.record.identity.concept_id == meta.expected_concept_id
        assert result.record.identity.native_id == meta.native_id
        assert result.record.identity.short_name == meta.short_name
        assert result.record.identity.version == meta.version
        assert "ScienceKeywords" not in discovery.model_dump_json()
    assert {path: path.read_bytes() for path in before} == before


def test_operation_metadata_encoding_client_state_and_run_collision_controls(
    tmp_path: Path,
) -> None:
    values = iter((0.0, 61.0))
    with _client(_ok) as client, pytest.raises(READMERetrievalError) as captured:
        retrieve_selected_readme(
            _selection(),
            client=client,
            resolver=lambda host: [PUBLIC4],
            artifact_directory=tmp_path / "time",
            monotonic=lambda: next(values),
        )
    assert captured.value.code == "README_OPERATION_TIMEOUT"

    metadata_settings = DatasetDocumentSettings(max_response_metadata_bytes=4)
    with pytest.raises(READMERetrievalError) as captured:
        _retrieve(tmp_path / "metadata", settings=metadata_settings)
    assert captured.value.code == "README_RESPONSE_METADATA_TOO_LARGE"

    def encoded(request):
        return httpx.Response(
            200,
            headers={"content-type": "application/pdf", "content-encoding": "gzip"},
            stream=httpx.ByteStream(PDF),
            request=request,
        )

    with pytest.raises(READMERetrievalError) as captured:
        _retrieve(tmp_path / "encoding", encoded)
    assert captured.value.code == "README_CONTENT_ENCODING"

    with (
        httpx.Client(
            transport=httpx.MockTransport(_ok), headers={"Authorization": "Bearer secret"}
        ) as client,
        pytest.raises(READMERetrievalError) as captured,
    ):
        retrieve_selected_readme(
            _selection(),
            client=client,
            resolver=lambda host: [PUBLIC4],
            artifact_directory=tmp_path / "auth",
        )
    assert captured.value.code == "README_CLIENT_STATE"

    collision_root = tmp_path / "collision"
    (collision_root / "fixed").mkdir(parents=True)
    with _client(_ok) as client, pytest.raises(READMERetrievalError) as captured:
        retrieve_selected_readme(
            _selection(),
            client=client,
            resolver=lambda host: [PUBLIC4],
            artifact_directory=collision_root,
            run_id_factory=lambda: "fixed",
        )
    assert captured.value.code == "README_ARTIFACT_EXISTS"


@pytest.mark.parametrize(
    "url",
    (
        "file:///tmp/readme.pdf",
        "https://-bad.example/readme.pdf",
        "https://bad_host.example/readme.pdf",
        "https://example.test:444/readme.pdf",
        "https://user:pass@example.test/readme.pdf",
        "https://example.test/readme.pdf#fragment",
    ),
)
def test_selected_url_policy_rejects_unsafe_destinations(tmp_path: Path, url: str) -> None:
    selected = _selection(url)
    with _client(_ok) as client, pytest.raises(READMERetrievalError):
        retrieve_selected_readme(
            selected,
            client=client,
            resolver=lambda host: [PUBLIC4],
            artifact_directory=tmp_path,
        )


def test_empty_invalid_dns_results_are_typed(tmp_path: Path) -> None:
    for answers, code in (([], "README_DNS_EMPTY"), (["not-an-ip"], "README_DNS_ADDRESS_INVALID")):
        with pytest.raises(READMERetrievalError) as captured:
            _retrieve(tmp_path / code, resolver=lambda host, values=answers: values)
        assert captured.value.code == code