Spaces:
Running on Zero
Running on Zero
Download src/gcmd_classifier/datasets/documents.py from igerasimov/GCMD_Keyword_Classifier_MVP: direct link, hf CLI and curl.
- Browser
- Download file 18.2 kB
-
https://huggingface.co/spaces/igerasimov/GCMD_Keyword_Classifier_MVP/resolve/main/src/gcmd_classifier/datasets/documents.py
- Command line
-
hf download hf://spaces/igerasimov/GCMD_Keyword_Classifier_MVP/src/gcmd_classifier/datasets/documents.py
-
curl -L -o documents.py https://huggingface.co/spaces/igerasimov/GCMD_Keyword_Classifier_MVP/resolve/main/src/gcmd_classifier/datasets/documents.py
18.2 kB
| """Selected-URL-only public README PDF retrieval with deterministic safety controls.""" | |
| from __future__ import annotations | |
| import hashlib | |
| import ipaddress | |
| import re | |
| import time | |
| from collections.abc import Callable, Iterable | |
| from contextlib import closing | |
| from dataclasses import dataclass | |
| from datetime import UTC, datetime | |
| from pathlib import Path | |
| from urllib.parse import parse_qsl, urljoin, urlsplit | |
| from uuid import uuid4 | |
| import httpx | |
| from gcmd_classifier.config import DatasetDocumentSettings | |
| from gcmd_classifier.datasets.errors import ( | |
| READMEDocumentError, | |
| READMEError, | |
| READMERetrievalError, | |
| READMESelectionError, | |
| ) | |
| from gcmd_classifier.datasets.models import ( | |
| DatasetIdentity, | |
| READMERetrievalRecord, | |
| RedirectRecord, | |
| SourceArtifactReference, | |
| ValidatedREADMESelection, | |
| ) | |
| from gcmd_classifier.datasets.readme_discovery import validate_selection_binding | |
| _REDIRECTS = {301, 302, 303, 307, 308} | |
| _AUTH_QUERY_NAMES = { | |
| "access_token", | |
| "api_key", | |
| "apikey", | |
| "auth", | |
| "authorization", | |
| "credential", | |
| "jwt", | |
| "password", | |
| "signature", | |
| "sig", | |
| "token", | |
| } | |
| _SAFE_HEADERS = {"content-type", "content-length", "content-disposition", "etag", "last-modified"} | |
| _PDF_HEADER = re.compile(rb"%PDF-[12]\.\d") | |
| _FILENAME = re.compile(r"filename\*?=(?:UTF-8''|\")?([^\";]+)", re.IGNORECASE) | |
| Resolver = Callable[[str], Iterable[str]] | |
| class RetrievedREADME: | |
| """Typed public result plus bounded exact bytes and internal artifact path.""" | |
| record: READMERetrievalRecord | |
| content: bytes | |
| def _retrieval_error(exc: httpx.HTTPError) -> None: | |
| if isinstance(exc, httpx.TimeoutException): | |
| raise READMERetrievalError("README_TIMEOUT", "README retrieval timed out.") from exc | |
| if isinstance(exc, httpx.TransportError): | |
| raise READMERetrievalError( | |
| "README_TRANSPORT_ERROR", "README connection, TLS, or transport failed." | |
| ) from exc | |
| raise READMERetrievalError("README_HTTP_ERROR", "README request failed.") from exc | |
| def _validated_public_url(url: str) -> tuple[str, str]: | |
| if not isinstance(url, str) or not url: | |
| raise READMERetrievalError("README_URL_INVALID", "README URL is invalid.") | |
| try: | |
| parsed = urlsplit(url) | |
| port = parsed.port | |
| except ValueError as exc: | |
| raise READMERetrievalError("README_URL_INVALID", "README URL is malformed.") from exc | |
| if parsed.scheme not in {"http", "https"}: | |
| raise READMERetrievalError("README_URL_SCHEME", "README URL scheme is not supported.") | |
| expected_port = 80 if parsed.scheme == "http" else 443 | |
| if port not in (None, expected_port): | |
| raise READMERetrievalError("README_URL_PORT", "README URL must use its default port.") | |
| if parsed.username is not None or parsed.password is not None: | |
| raise READMERetrievalError("README_URL_USERINFO", "README URL contains credentials.") | |
| if parsed.fragment: | |
| raise READMERetrievalError("README_URL_FRAGMENT", "README URL fragments are prohibited.") | |
| host = parsed.hostname | |
| if not host or any(character.isspace() for character in host): | |
| raise READMERetrievalError("README_URL_HOST", "README URL host is malformed.") | |
| try: | |
| ascii_host = host.encode("idna").decode("ascii") | |
| except UnicodeError as exc: | |
| raise READMERetrievalError("README_URL_HOST", "README URL host is malformed.") from exc | |
| try: | |
| ipaddress.ip_address(ascii_host) | |
| except ValueError: | |
| labels = ascii_host.rstrip(".").split(".") | |
| if ( | |
| len(ascii_host) > 253 | |
| or not labels | |
| or any( | |
| not label | |
| or len(label) > 63 | |
| or label.startswith("-") | |
| or label.endswith("-") | |
| or re.fullmatch(r"[A-Za-z0-9-]+", label) is None | |
| for label in labels | |
| ) | |
| ): | |
| raise READMERetrievalError("README_URL_HOST", "README URL host is malformed.") from None | |
| for name, _ in parse_qsl(parsed.query, keep_blank_values=True): | |
| normalized = name.strip().lower().replace("-", "_") | |
| if normalized in _AUTH_QUERY_NAMES or any( | |
| marker in normalized for marker in ("token", "password", "credential", "authorization") | |
| ): | |
| raise READMERetrievalError( | |
| "README_URL_AUTH_QUERY", "README URL contains authentication data." | |
| ) | |
| return url, host | |
| def _resolve_public(host: str, resolver: Resolver) -> tuple[str, ...]: | |
| try: | |
| values = tuple(str(value) for value in resolver(host)) | |
| except Exception as exc: | |
| raise READMERetrievalError("README_DNS_FAILURE", "README host resolution failed.") from exc | |
| if not values: | |
| raise READMERetrievalError("README_DNS_EMPTY", "README host resolved to no addresses.") | |
| for value in values: | |
| try: | |
| address = ipaddress.ip_address(value) | |
| except ValueError as exc: | |
| raise READMERetrievalError( | |
| "README_DNS_ADDRESS_INVALID", "README host resolved to an invalid address." | |
| ) from exc | |
| if ( | |
| not address.is_global | |
| or address.is_multicast | |
| or address.is_reserved | |
| or address.is_unspecified | |
| or address.is_loopback | |
| or address.is_link_local | |
| or address.is_private | |
| ): | |
| raise READMERetrievalError( | |
| "README_NON_PUBLIC_ADDRESS", "README destination is not globally routable." | |
| ) | |
| return values | |
| def _looks_like_login(content_type: str | None, body: bytes) -> bool: | |
| sample = body[:16384].lower() | |
| if "html" not in (content_type or "").lower() and not sample.lstrip().startswith(b"<"): | |
| return False | |
| return any( | |
| marker in sample | |
| for marker in ( | |
| b"<form", | |
| b'type="password"', | |
| b"type='password'", | |
| b"sign in", | |
| b"log in", | |
| b"login", | |
| ) | |
| ) | |
| def _login_target(url: str) -> bool: | |
| path = urlsplit(url).path.lower() | |
| return any(part in path for part in ("/login", "/log-in", "/signin", "/sign-in", "/auth")) | |
| def validate_pdf_bytes(body: bytes, declared_content_type: str | None) -> tuple[str, str]: | |
| """Validate supported PDF bytes without extracting or interpreting content.""" | |
| media_type = (declared_content_type or "").split(";", 1)[0].strip().lower() | |
| if _looks_like_login(declared_content_type, body): | |
| raise READMERetrievalError("README_NOT_PUBLIC", "README returned a login page.") | |
| if not body: | |
| raise READMEDocumentError("UNSUPPORTED_README_FORMAT", "README response is empty.") | |
| if media_type in {"text/html", "application/xhtml+xml"}: | |
| raise READMEDocumentError("UNSUPPORTED_README_FORMAT", "README response is HTML.") | |
| header = _PDF_HEADER.search(body[:1024]) | |
| eof_position = body.rstrip().rfind(b"%%EOF") | |
| structurally_valid = ( | |
| header is not None | |
| and b"startxref" in body[-4096:] | |
| and eof_position >= len(body.rstrip()) - len(b"%%EOF") | |
| and b"obj" in body | |
| ) | |
| if not structurally_valid: | |
| raise READMEDocumentError( | |
| "UNSUPPORTED_README_FORMAT", "README bytes are not a structurally supported PDF." | |
| ) | |
| allowed_types = {"", "application/pdf", "application/octet-stream", "binary/octet-stream"} | |
| if media_type not in allowed_types: | |
| raise READMEDocumentError( | |
| "UNSUPPORTED_README_FORMAT", "README response metadata declares an unsupported type." | |
| ) | |
| return "application/pdf", "pdf-header-object-startxref-eof-v1" | |
| def _safe_metadata(headers: httpx.Headers, limit: int) -> dict[str, str]: | |
| selected = { | |
| key.lower(): value for key, value in headers.items() if key.lower() in _SAFE_HEADERS | |
| } | |
| if sum(len(key) + len(value) for key, value in selected.items()) > limit: | |
| raise READMERetrievalError( | |
| "README_RESPONSE_METADATA_TOO_LARGE", "README response metadata exceeds its limit." | |
| ) | |
| return selected | |
| def _server_filename(content_disposition: str | None) -> str | None: | |
| if not content_disposition or not (match := _FILENAME.search(content_disposition)): | |
| return None | |
| filename = Path(match.group(1).strip()).name | |
| filename = "".join(character for character in filename if character.isprintable()) | |
| return filename[:255] or None | |
| def retrieve_selected_readme( | |
| selection: ValidatedREADMESelection, | |
| *, | |
| client: httpx.Client, | |
| resolver: Resolver, | |
| artifact_directory: Path, | |
| settings: DatasetDocumentSettings | None = None, | |
| utc_now: Callable[[], datetime] | None = None, | |
| monotonic: Callable[[], float] | None = None, | |
| run_id_factory: Callable[[], str] | None = None, | |
| ) -> RetrievedREADME: | |
| """Retrieve only an exact, explicitly selected CMR-discovered README URL.""" | |
| if not isinstance(selection, ValidatedREADMESelection): | |
| raise READMESelectionError( | |
| "README_SELECTION_REQUIRED", "A typed validated README selection is required." | |
| ) | |
| validate_selection_binding(selection) | |
| policy = settings or DatasetDocumentSettings() | |
| current_url, _ = _validated_public_url(selection.selected_url) | |
| now = utc_now or (lambda: datetime.now(UTC)) | |
| elapsed = monotonic or time.monotonic | |
| started = elapsed() | |
| redirects: list[RedirectRecord] = [] | |
| visited = {current_url} | |
| timeout = httpx.Timeout( | |
| connect=policy.connect_timeout_seconds, | |
| read=policy.read_timeout_seconds, | |
| write=policy.read_timeout_seconds, | |
| pool=policy.connect_timeout_seconds, | |
| ) | |
| run_id = (run_id_factory or (lambda: uuid4().hex))() | |
| if not re.fullmatch(r"[A-Za-z0-9_-]{1,64}", run_id): | |
| raise READMERetrievalError("README_RUN_ID_INVALID", "README artifact run ID is unsafe.") | |
| run_directory = artifact_directory / run_id | |
| try: | |
| run_directory.mkdir(parents=True, exist_ok=False) | |
| except FileExistsError as exc: | |
| raise READMERetrievalError( | |
| "README_ARTIFACT_EXISTS", | |
| "README artifact location already exists; overwrite is prohibited.", | |
| ) from exc | |
| partial_path = run_directory / "document.partial" | |
| while True: | |
| if elapsed() - started > policy.operation_timeout_seconds: | |
| raise READMERetrievalError( | |
| "README_OPERATION_TIMEOUT", "README retrieval exceeded its overall time limit." | |
| ) | |
| _, host = _validated_public_url(current_url) | |
| first_resolution = _resolve_public(host, resolver) | |
| second_resolution = _resolve_public(host, resolver) | |
| validated_addresses = tuple(dict.fromkeys((*first_resolution, *second_resolution))) | |
| try: | |
| request = client.build_request( | |
| "GET", | |
| current_url, | |
| headers={ | |
| "Accept": "application/pdf,application/octet-stream;q=0.8", | |
| "Accept-Encoding": "identity", | |
| }, | |
| timeout=timeout, | |
| ) | |
| if "authorization" in request.headers or "cookie" in request.headers: | |
| raise READMERetrievalError( | |
| "README_CLIENT_STATE", | |
| "README requests must not contain credentials or cookies.", | |
| ) | |
| with closing( | |
| client.send(request, stream=True, follow_redirects=False, auth=None) | |
| ) as response: | |
| status = response.status_code | |
| if status in _REDIRECTS: | |
| location = response.headers.get("location") | |
| if not location: | |
| raise READMERetrievalError( | |
| "README_REDIRECT_LOCATION", "README redirect lacks a Location header." | |
| ) | |
| target = urljoin(current_url, location) | |
| _validated_public_url(target) | |
| if _login_target(target): | |
| raise READMERetrievalError( | |
| "README_NOT_PUBLIC", "README redirected to authentication." | |
| ) | |
| if target in visited: | |
| raise READMERetrievalError( | |
| "README_REDIRECT_LOOP", "README redirect loop detected." | |
| ) | |
| if len(redirects) >= policy.max_redirects: | |
| raise READMERetrievalError( | |
| "README_REDIRECT_LIMIT", "README redirect limit exceeded." | |
| ) | |
| redirects.append( | |
| RedirectRecord( | |
| status_code=status, source_url=current_url, target_url=target | |
| ) | |
| ) | |
| visited.add(target) | |
| current_url = target | |
| continue | |
| if 300 <= status < 400: | |
| raise READMERetrievalError( | |
| "README_REDIRECT_UNSUPPORTED", "README returned an unsupported redirect." | |
| ) | |
| if status in {401, 403}: | |
| raise READMERetrievalError( | |
| "README_NOT_PUBLIC", "README is not publicly accessible." | |
| ) | |
| if not 200 <= status < 300: | |
| raise READMERetrievalError( | |
| "README_HTTP_STATUS", f"README returned HTTP status {status}." | |
| ) | |
| metadata = _safe_metadata(response.headers, policy.max_response_metadata_bytes) | |
| content_length = response.headers.get("content-length") | |
| declared_length: int | None = None | |
| if content_length is not None: | |
| try: | |
| declared_length = int(content_length) | |
| except ValueError as exc: | |
| raise READMERetrievalError( | |
| "README_CONTENT_LENGTH", "README Content-Length is invalid." | |
| ) from exc | |
| if declared_length < 0 or declared_length > policy.max_download_bytes: | |
| raise READMERetrievalError( | |
| "README_TOO_LARGE", "README exceeds the configured size limit." | |
| ) | |
| encoding = response.headers.get("content-encoding") | |
| if encoding and encoding.lower() != "identity": | |
| raise READMERetrievalError( | |
| "README_CONTENT_ENCODING", "README used unsupported content encoding." | |
| ) | |
| chunks: list[bytes] = [] | |
| byte_count = 0 | |
| response_chunks = ( | |
| (response.content,) if response.is_stream_consumed else response.iter_raw() | |
| ) | |
| with partial_path.open("wb") as artifact_file: | |
| for chunk in response_chunks: | |
| if elapsed() - started > policy.operation_timeout_seconds: | |
| raise READMERetrievalError( | |
| "README_OPERATION_TIMEOUT", | |
| "README retrieval exceeded its overall time limit.", | |
| ) | |
| byte_count += len(chunk) | |
| if byte_count > policy.max_download_bytes: | |
| raise READMERetrievalError( | |
| "README_TOO_LARGE", "README exceeds the configured size limit." | |
| ) | |
| artifact_file.write(chunk) | |
| chunks.append(chunk) | |
| if declared_length is not None and declared_length != byte_count: | |
| raise READMERetrievalError( | |
| "README_TRUNCATED", "README byte count differs from Content-Length." | |
| ) | |
| body = b"".join(chunks) | |
| media_type, validation_method = validate_pdf_bytes( | |
| body, response.headers.get("content-type") | |
| ) | |
| final_path = run_directory / "document.pdf" | |
| partial_path.replace(final_path) | |
| digest = hashlib.sha256(body).hexdigest() | |
| retrieved_at = now().astimezone(UTC).isoformat().replace("+00:00", "Z") | |
| identity = DatasetIdentity( | |
| concept_id=selection.concept_id, | |
| native_id=selection.native_id, | |
| short_name=selection.short_name, | |
| version=selection.version, | |
| cmr_revision_id=selection.cmr_revision_id, | |
| ) | |
| artifact = SourceArtifactReference( | |
| artifact_id=f"readme-{digest}", | |
| sha256=digest, | |
| byte_size=byte_count, | |
| media_type=media_type, | |
| storage_reference=f"{run_id}/document.pdf", | |
| ) | |
| record = READMERetrievalRecord( | |
| identity=identity, | |
| cmr_source_sha256=selection.cmr_source_sha256, | |
| selected_candidate_id=selection.candidate_id, | |
| selected_source_index=selection.source_index, | |
| selected_source_entry=selection.source_entry, | |
| submitted_url=selection.selected_url, | |
| final_url=current_url, | |
| redirects=tuple(redirects), | |
| artifact=artifact, | |
| retrieved_at=retrieved_at, | |
| status_code=status, | |
| response_metadata=metadata, | |
| server_filename=_server_filename(response.headers.get("content-disposition")), | |
| public_address_validation="all-resolved-addresses-global-v1:" | |
| + ",".join(validated_addresses), | |
| document_validation_method=validation_method, | |
| ) | |
| return RetrievedREADME(record=record, content=body) | |
| except READMEError: | |
| raise | |
| except httpx.HTTPError as exc: | |
| _retrieval_error(exc) | |