Spaces:
Running on Zero
Running on Zero
Download tests/datasets/test_documents.py from igerasimov/GCMD_Keyword_Classifier_MVP: direct link, hf CLI and curl.
- Browser
- Download file 14.7 kB
-
https://huggingface.co/spaces/igerasimov/GCMD_Keyword_Classifier_MVP/resolve/main/tests/datasets/test_documents.py
- Command line
-
hf download hf://spaces/igerasimov/GCMD_Keyword_Classifier_MVP/tests/datasets/test_documents.py
-
curl -L -o test_documents.py https://huggingface.co/spaces/igerasimov/GCMD_Keyword_Classifier_MVP/resolve/main/tests/datasets/test_documents.py
14.7 kB
| from __future__ import annotations | |
| import hashlib | |
| import json | |
| from datetime import UTC, datetime | |
| from pathlib import Path | |
| import httpx | |
| import pytest | |
| from gcmd_classifier.config import DatasetDocumentSettings | |
| from gcmd_classifier.datasets.cmr import resolve_cmr_collection, validate_cmr_url | |
| from gcmd_classifier.datasets.documents import retrieve_selected_readme, validate_pdf_bytes | |
| from gcmd_classifier.datasets.errors import ( | |
| READMEDocumentError, | |
| READMERetrievalError, | |
| READMESelectionError, | |
| ) | |
| from gcmd_classifier.datasets.models import BlindCollectionView, CMRRetrievedSource, DatasetIdentity | |
| from gcmd_classifier.datasets.readme_discovery import ( | |
| discover_readme_candidates, | |
| select_readme_candidate, | |
| ) | |
| from tests.datasets.fixture_manifest import load_fixture_manifest | |
| URL = "https://docs.example.test/readme.pdf" | |
| NOW = datetime(2026, 8, 13, 13, 0, tzinfo=UTC) | |
| PDF = b"%PDF-1.7\n1 0 obj\n<<>>\nendobj\nstartxref\n9\n%%EOF\n" | |
| PUBLIC4 = "93.184.216.34" | |
| PUBLIC6 = "2606:2800:220:1:248:1893:25c8:1946" | |
| def _selection(url: str = URL): | |
| view = BlindCollectionView( | |
| identity=DatasetIdentity( | |
| concept_id="C1-P", | |
| native_id="TARGET_001", | |
| short_name="TARGET", | |
| version="001", | |
| cmr_revision_id=4, | |
| ), | |
| derived_native_id="TARGET_001", | |
| related_urls=({"Subtype": "READ-ME", "URL": url, "Description": "README"},), | |
| blind_view_sha256="a" * 64, | |
| ) | |
| discovery = discover_readme_candidates(view, cmr_source_sha256="b" * 64) | |
| return select_readme_candidate(discovery, [(0, url)], utc_now=lambda: NOW) | |
| def _client(handler) -> httpx.Client: | |
| return httpx.Client(transport=httpx.MockTransport(handler), follow_redirects=False) | |
| def _ok( | |
| request: httpx.Request, | |
| *, | |
| body: bytes = PDF, | |
| content_type: str = "application/pdf", | |
| headers=None, | |
| ): | |
| return httpx.Response( | |
| 200, | |
| headers={"content-type": content_type, **(headers or {})}, | |
| content=body, | |
| request=request, | |
| ) | |
| def _retrieve( | |
| tmp_path: Path, handler=_ok, *, resolver=lambda host: [PUBLIC4], settings=None, selection=None | |
| ): | |
| with _client(handler) as client: | |
| return retrieve_selected_readme( | |
| selection or _selection(), | |
| client=client, | |
| resolver=resolver, | |
| artifact_directory=tmp_path, | |
| settings=settings, | |
| utc_now=lambda: NOW, | |
| ) | |
| def test_public_ipv4_ipv6_and_mixed_public_addresses_are_allowed( | |
| tmp_path: Path, addresses: list[str] | |
| ) -> None: | |
| result = _retrieve(tmp_path, resolver=lambda host: addresses) | |
| assert result.content == PDF | |
| assert result.record.artifact.sha256 == hashlib.sha256(PDF).hexdigest() | |
| storage_reference = result.record.artifact.storage_reference | |
| assert storage_reference is not None | |
| assert (tmp_path / storage_reference).read_bytes() == PDF | |
| assert not Path(storage_reference).is_absolute() | |
| def test_non_public_destinations_are_rejected_without_http(tmp_path: Path, address: str) -> None: | |
| calls = 0 | |
| def handler(request): | |
| nonlocal calls | |
| calls += 1 | |
| return _ok(request) | |
| with pytest.raises(READMERetrievalError) as captured: | |
| _retrieve(tmp_path, handler, resolver=lambda host: [address]) | |
| assert captured.value.code == "README_NON_PUBLIC_ADDRESS" | |
| assert calls == 0 | |
| def test_mixed_public_nonpublic_and_dns_rebinding_are_rejected(tmp_path: Path) -> None: | |
| with pytest.raises(READMERetrievalError): | |
| _retrieve(tmp_path, resolver=lambda host: [PUBLIC4, "10.0.0.1"]) | |
| answers = iter(([PUBLIC4], ["127.0.0.1"])) | |
| with pytest.raises(READMERetrievalError): | |
| _retrieve(tmp_path, resolver=lambda host: next(answers)) | |
| def test_valid_redirect_revalidates_destination_and_preserves_history(tmp_path: Path) -> None: | |
| target = "https://cdn.example.test/file" | |
| requests = [] | |
| def handler(request): | |
| requests.append(str(request.url)) | |
| return ( | |
| httpx.Response(302, headers={"location": target}, request=request) | |
| if len(requests) == 1 | |
| else _ok(request) | |
| ) | |
| result = _retrieve(tmp_path, handler) | |
| assert requests == [URL, target] | |
| assert result.record.final_url == target | |
| assert result.record.redirects[0].target_url == target | |
| def test_unsafe_and_authentication_redirects_are_rejected(tmp_path: Path, target: str) -> None: | |
| def handler(request): | |
| return httpx.Response(302, headers={"location": target}, request=request) | |
| with pytest.raises(READMERetrievalError): | |
| _retrieve(tmp_path, handler) | |
| def test_redirect_loop_and_limit(tmp_path: Path) -> None: | |
| second = "https://docs.example.test/second.pdf" | |
| def loop(request): | |
| target = second if str(request.url) == URL else URL | |
| return httpx.Response(302, headers={"location": target}, request=request) | |
| with pytest.raises(READMERetrievalError) as captured: | |
| _retrieve(tmp_path, loop) | |
| assert captured.value.code == "README_REDIRECT_LOOP" | |
| with pytest.raises(READMERetrievalError) as captured: | |
| _retrieve(tmp_path, loop, settings=DatasetDocumentSettings(max_redirects=0)) | |
| assert captured.value.code == "README_REDIRECT_LIMIT" | |
| def test_authentication_status_is_not_public(tmp_path: Path, status: int) -> None: | |
| with pytest.raises(READMERetrievalError) as captured: | |
| _retrieve(tmp_path, lambda request: httpx.Response(status, request=request)) | |
| assert captured.value.code == "README_NOT_PUBLIC" | |
| def test_login_html_is_not_public(tmp_path: Path) -> None: | |
| with pytest.raises(READMERetrievalError) as captured: | |
| _retrieve( | |
| tmp_path, | |
| lambda request: _ok( | |
| request, | |
| body=b"<html><form><input type='password'></form></html>", | |
| content_type="text/html", | |
| ), | |
| ) | |
| assert captured.value.code == "README_NOT_PUBLIC" | |
| def test_transport_failures_are_typed_and_preserve_no_false_success( | |
| tmp_path: Path, exc: Exception | |
| ) -> None: | |
| def handler(request): | |
| raise exc | |
| with pytest.raises(READMERetrievalError): | |
| _retrieve(tmp_path, handler) | |
| def test_size_content_length_truncation_and_operation_limits(tmp_path: Path) -> None: | |
| settings = DatasetDocumentSettings(max_download_bytes=10) | |
| with pytest.raises(READMERetrievalError) as captured: | |
| _retrieve(tmp_path, settings=settings) | |
| assert captured.value.code == "README_TOO_LARGE" | |
| def truncated(request): | |
| return _ok(request, headers={"content-length": str(len(PDF) + 1)}) | |
| with pytest.raises(READMERetrievalError) as captured: | |
| _retrieve(tmp_path, truncated) | |
| assert captured.value.code == "README_TRUNCATED" | |
| def test_unsupported_and_misleading_content_is_rejected(body: bytes, media: str) -> None: | |
| with pytest.raises(READMEDocumentError): | |
| validate_pdf_bytes(body, media) | |
| def test_pdf_byte_validation_is_extension_independent_and_stable(media: str | None) -> None: | |
| assert validate_pdf_bytes(PDF, media)[0] == "application/pdf" | |
| def test_raw_url_local_path_and_tampered_selection_cannot_enter_retriever(tmp_path: Path) -> None: | |
| with _client(_ok) as client, pytest.raises(READMESelectionError): | |
| retrieve_selected_readme( | |
| URL, client=client, resolver=lambda host: [PUBLIC4], artifact_directory=tmp_path | |
| ) | |
| selected = _selection() | |
| changed = selected.model_copy(update={"selected_url": "tests/fixtures/datasets/pdf/file.pdf"}) | |
| with _client(_ok) as client, pytest.raises(READMESelectionError): | |
| retrieve_selected_readme( | |
| changed, client=client, resolver=lambda host: [PUBLIC4], artifact_directory=tmp_path | |
| ) | |
| def test_failure_preserves_partial_artifact_and_performs_no_cleanup(tmp_path: Path) -> None: | |
| with pytest.raises(READMEDocumentError): | |
| _retrieve( | |
| tmp_path, | |
| lambda request: _ok(request, body=b"not a pdf", content_type="application/pdf"), | |
| ) | |
| partials = list(tmp_path.rglob("document.partial")) | |
| assert len(partials) == 1 and partials[0].read_bytes() == b"not a pdf" | |
| def test_all_frozen_cases_require_discovery_selection_and_exact_mocked_bytes( | |
| tmp_path: Path, | |
| ) -> None: | |
| manifest, cases = load_fixture_manifest(Path("tests/fixtures/datasets/cases.json")) | |
| before = { | |
| path: path.read_bytes() for case in cases for path in (case.cmr_path, case.readme_path) | |
| } | |
| assert len(manifest.cases) == 4 | |
| for case in cases: | |
| meta = case.metadata | |
| cmr_bytes = case.cmr_path.read_bytes() | |
| pdf_bytes = case.readme_path.read_bytes() | |
| validated = validate_cmr_url(meta.cmr_url) | |
| source = CMRRetrievedSource( | |
| submitted_url=meta.cmr_url, | |
| final_url=meta.cmr_url, | |
| retrieved_at="2026-08-13T00:00:00Z", | |
| status_code=200, | |
| response_bytes=cmr_bytes, | |
| source_text=cmr_bytes.decode(), | |
| sha256=hashlib.sha256(cmr_bytes).hexdigest(), | |
| parsed_json=json.loads(cmr_bytes), | |
| ) | |
| resolved = resolve_cmr_collection(validated, source) | |
| discovery = discover_readme_candidates(resolved.blind_view, cmr_source_sha256=source.sha256) | |
| matching = [c for c in discovery.candidates if c.url == meta.selected_readme_url] | |
| assert len(matching) == 1 | |
| selected = select_readme_candidate( | |
| discovery, [(matching[0].source_index, matching[0].url)], utc_now=lambda: NOW | |
| ) | |
| def exact_transport(request, expected=meta.selected_readme_url, content=pdf_bytes): | |
| assert str(request.url) == expected | |
| return _ok(request, body=content) | |
| result = _retrieve(tmp_path / meta.case_id, exact_transport, selection=selected) | |
| assert result.content == pdf_bytes | |
| assert result.record.artifact.sha256 == meta.readme_sha256 | |
| assert result.record.identity.concept_id == meta.expected_concept_id | |
| assert result.record.identity.native_id == meta.native_id | |
| assert result.record.identity.short_name == meta.short_name | |
| assert result.record.identity.version == meta.version | |
| assert "ScienceKeywords" not in discovery.model_dump_json() | |
| assert {path: path.read_bytes() for path in before} == before | |
| def test_operation_metadata_encoding_client_state_and_run_collision_controls( | |
| tmp_path: Path, | |
| ) -> None: | |
| values = iter((0.0, 61.0)) | |
| with _client(_ok) as client, pytest.raises(READMERetrievalError) as captured: | |
| retrieve_selected_readme( | |
| _selection(), | |
| client=client, | |
| resolver=lambda host: [PUBLIC4], | |
| artifact_directory=tmp_path / "time", | |
| monotonic=lambda: next(values), | |
| ) | |
| assert captured.value.code == "README_OPERATION_TIMEOUT" | |
| metadata_settings = DatasetDocumentSettings(max_response_metadata_bytes=4) | |
| with pytest.raises(READMERetrievalError) as captured: | |
| _retrieve(tmp_path / "metadata", settings=metadata_settings) | |
| assert captured.value.code == "README_RESPONSE_METADATA_TOO_LARGE" | |
| def encoded(request): | |
| return httpx.Response( | |
| 200, | |
| headers={"content-type": "application/pdf", "content-encoding": "gzip"}, | |
| stream=httpx.ByteStream(PDF), | |
| request=request, | |
| ) | |
| with pytest.raises(READMERetrievalError) as captured: | |
| _retrieve(tmp_path / "encoding", encoded) | |
| assert captured.value.code == "README_CONTENT_ENCODING" | |
| with ( | |
| httpx.Client( | |
| transport=httpx.MockTransport(_ok), headers={"Authorization": "Bearer secret"} | |
| ) as client, | |
| pytest.raises(READMERetrievalError) as captured, | |
| ): | |
| retrieve_selected_readme( | |
| _selection(), | |
| client=client, | |
| resolver=lambda host: [PUBLIC4], | |
| artifact_directory=tmp_path / "auth", | |
| ) | |
| assert captured.value.code == "README_CLIENT_STATE" | |
| collision_root = tmp_path / "collision" | |
| (collision_root / "fixed").mkdir(parents=True) | |
| with _client(_ok) as client, pytest.raises(READMERetrievalError) as captured: | |
| retrieve_selected_readme( | |
| _selection(), | |
| client=client, | |
| resolver=lambda host: [PUBLIC4], | |
| artifact_directory=collision_root, | |
| run_id_factory=lambda: "fixed", | |
| ) | |
| assert captured.value.code == "README_ARTIFACT_EXISTS" | |
| def test_selected_url_policy_rejects_unsafe_destinations(tmp_path: Path, url: str) -> None: | |
| selected = _selection(url) | |
| with _client(_ok) as client, pytest.raises(READMERetrievalError): | |
| retrieve_selected_readme( | |
| selected, | |
| client=client, | |
| resolver=lambda host: [PUBLIC4], | |
| artifact_directory=tmp_path, | |
| ) | |
| def test_empty_invalid_dns_results_are_typed(tmp_path: Path) -> None: | |
| for answers, code in (([], "README_DNS_EMPTY"), (["not-an-ip"], "README_DNS_ADDRESS_INVALID")): | |
| with pytest.raises(READMERetrievalError) as captured: | |
| _retrieve(tmp_path / code, resolver=lambda host, values=answers: values) | |
| assert captured.value.code == code | |