Spaces:
Running on Zero
Running on Zero
Download src/gcmd_classifier/datasets/models.py from igerasimov/GCMD_Keyword_Classifier_MVP: direct link, hf CLI and curl.
- Browser
- Download file 67.4 kB
-
https://huggingface.co/spaces/igerasimov/GCMD_Keyword_Classifier_MVP/resolve/main/src/gcmd_classifier/datasets/models.py
- Command line
-
hf download hf://spaces/igerasimov/GCMD_Keyword_Classifier_MVP/src/gcmd_classifier/datasets/models.py
-
curl -L -o models.py https://huggingface.co/spaces/igerasimov/GCMD_Keyword_Classifier_MVP/resolve/main/src/gcmd_classifier/datasets/models.py
67.4 kB
| """Dataset-only typed contracts for the Dataset README Classifier.""" | |
| from __future__ import annotations | |
| from enum import Enum | |
| from typing import Any, Literal | |
| from pydantic import BaseModel, ConfigDict, Field, model_validator | |
| from gcmd_classifier.models import HierarchyLevel | |
| class DatasetProcessingStatus(str, Enum): # noqa: UP042 | |
| """Execution state for one dataset workflow.""" | |
| COMPLETED = "completed" | |
| PARTIAL = "partial" | |
| FAILED = "failed" | |
| SKIPPED = "skipped" | |
| class DatasetClassificationOutcome(str, Enum): # noqa: UP042 | |
| """Semantic dataset outcome after classification is attempted.""" | |
| CLASSIFIED = "classified" | |
| PENDING_REVIEW = "pending_review" | |
| NOT_CLASSIFIED = "not_classified" | |
| class DatasetReviewStatus(str, Enum): # noqa: UP042 | |
| """Human-review state for a dataset result.""" | |
| NOT_REQUIRED = "not_required" | |
| PENDING = "pending" | |
| COMPLETED = "completed" | |
| class DatasetClassificationFinalStatus(str, Enum): # noqa: UP042 | |
| """Automated status for one dataset classification.""" | |
| ACCEPTED = "accepted" | |
| REDUCED_TO_ANCESTOR = "reduced_to_ancestor" | |
| REVIEW_REQUIRED = "review_required" | |
| REJECTED = "rejected" | |
| class DatasetStage(str, Enum): # noqa: UP042 | |
| """Dataset workflow stages used by diagnostics.""" | |
| SOURCE = "source" | |
| README_DISCOVERY = "readme_discovery" | |
| README_RETRIEVAL = "readme_retrieval" | |
| EXTRACTION = "extraction" | |
| PRODUCT_RESOLUTION = "product_resolution" | |
| EVIDENCE_SELECTION = "evidence_selection" | |
| CLASSIFICATION = "classification" | |
| VALIDATION = "validation" | |
| BLIND_PERSISTENCE = "blind_persistence" | |
| COMPARISON = "comparison" | |
| class CoverageMode(str, Enum): # noqa: UP042 | |
| """Supported README document coverage modes.""" | |
| SINGLE_PRODUCT = "single_product" | |
| EXPLICIT_MULTI_PRODUCT = "explicit_multi_product" | |
| PRODUCT_FAMILY = "product_family" | |
| VAGUE_PRODUCT_SCOPE = "vague_product_scope" | |
| UNDETERMINED = "undetermined" | |
| class ProductScopeType(str, Enum): # noqa: UP042 | |
| """How an evidence block applies to products.""" | |
| EXACT_TARGET = "exact_target" | |
| ENUMERATED_TARGET = "enumerated_target" | |
| PRODUCT_FAMILY = "product_family" | |
| AMBIGUOUS = "ambiguous" | |
| OTHER_PRODUCT = "other_product" | |
| DOCUMENT_CONTEXT = "document_context" | |
| MIXED_PRODUCT_TRANSITION = "mixed_product_transition" | |
| class EvidenceEligibility(str, Enum): # noqa: UP042 | |
| """Product applicability eligibility before evidence selection.""" | |
| ELIGIBLE = "eligible" | |
| EXCLUDED = "excluded" | |
| REVIEW_ONLY = "review_only" | |
| DOCUMENT_CONTEXT = "document_context" | |
| class EvidenceRole(str, Enum): # noqa: UP042 | |
| """Scientific or contextual role of one source evidence block.""" | |
| PRODUCT_IDENTITY = "product_identity" | |
| SCIENTIFIC_DESCRIPTION = "scientific_description" | |
| MEASURED_VARIABLE = "measured_variable" | |
| GEOPHYSICAL_PARAMETER = "geophysical_parameter" | |
| METHOD_OR_ALGORITHM = "method_or_algorithm" | |
| SPATIAL_TEMPORAL_CONTEXT = "spatial_temporal_context" | |
| PROCESSING_OR_FORMAT = "processing_or_format" | |
| QUALIFICATION = "qualification" | |
| OTHER = "other" | |
| class ComparisonCategory(str, Enum): # noqa: UP042 | |
| """Post-inference relationship between blind and expert assignments.""" | |
| ALREADY_ASSIGNED = "already_assigned" | |
| PROPOSED_ADDITION = "proposed_addition" | |
| PROPOSED_REFINEMENT = "proposed_refinement" | |
| REDUNDANT_RESULT = "redundant_result" | |
| REVIEW_REQUIRED = "review_required" | |
| class PageExtractionStatus(str, Enum): # noqa: UP042 | |
| """Deterministic page extraction outcome.""" | |
| EXTRACTED = "extracted" | |
| BLANK = "blank" | |
| IMAGE_ONLY = "image_only" | |
| class HeadingKind(str, Enum): # noqa: UP042 | |
| """Literal or deterministic fallback heading kind.""" | |
| NUMBERED = "numbered" | |
| ALL_CAPS = "all_caps" | |
| COLON_LABEL = "colon_label" | |
| DERIVED_DOCUMENT = "derived_document" | |
| class SectionLabelSource(str, Enum): # noqa: UP042 | |
| """Whether section metadata is literal source text or derived.""" | |
| LITERAL = "literal" | |
| DERIVED = "derived" | |
| NONE = "none" | |
| class DatasetModel(BaseModel): | |
| """Strict immutable base for serialized dataset contracts.""" | |
| model_config = ConfigDict(extra="forbid", frozen=True) | |
| class DatasetWarning(DatasetModel): | |
| """Non-fatal dataset workflow diagnostic.""" | |
| code: str = Field(min_length=1) | |
| message: str = Field(min_length=1) | |
| stage: DatasetStage | None = None | |
| details: dict[str, Any] | None = None | |
| class DatasetErrorRecord(DatasetModel): | |
| """Structured dataset workflow failure.""" | |
| code: str = Field(min_length=1) | |
| message: str = Field(min_length=1) | |
| stage: DatasetStage | |
| retry_count: int | None = Field(default=None, ge=0) | |
| details: dict[str, Any] | None = None | |
| class DatasetHierarchyAttemptDiagnostic(DatasetModel): | |
| """Sanitized facts for one failed dataset hierarchy model attempt.""" | |
| pipeline_stage: Literal["classification"] = "classification" | |
| model_stage: Literal["dataset_topic", "dataset_term", "dataset_variable"] | |
| attempt_number: int = Field(ge=1) | |
| retryable: bool | |
| failure_category: Literal[ | |
| "model_non_retryable", | |
| "model_retryable", | |
| "structured_response", | |
| "response_schema", | |
| "deterministic_validation", | |
| "internal", | |
| ] | |
| public_error_code: str = Field(min_length=1) | |
| provider: str = Field(min_length=1) | |
| model_name: str = Field(min_length=1) | |
| provider_http_status: int | None = Field(default=None, ge=100, le=599) | |
| provider_error_type: str | None = Field(default=None, max_length=128) | |
| provider_error_code: str | None = Field(default=None, max_length=128) | |
| provider_error_parameter: str | None = Field(default=None, max_length=128) | |
| provider_message: str | None = Field(default=None, max_length=256) | |
| provider_request_id: str | None = Field(default=None, max_length=256) | |
| packet_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| parent_uuid: str | None = None | |
| parent_name: str | None = None | |
| parent_level: str | None = None | |
| parent_canonical_path: str | None = None | |
| branch_id: str = Field(min_length=1) | |
| candidate_count: int = Field(ge=0) | |
| candidate_ids: tuple[str, ...] = Field(default_factory=tuple) | |
| candidate_ids_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| response_schema_name: str = Field(min_length=1) | |
| prompt_version: str = Field(min_length=1) | |
| prompt_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| prompt_character_count: int = Field(ge=0) | |
| estimated_prompt_tokens: int = Field(ge=0) | |
| elapsed_seconds: float = Field(ge=0.0) | |
| parsed_response_received: bool | |
| structured_validation_code: str | None = Field(default=None, max_length=128) | |
| class SourceArtifactReference(DatasetModel): | |
| """Immutable reference to exact source bytes.""" | |
| artifact_id: str = Field(min_length=1) | |
| sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| byte_size: int = Field(ge=0) | |
| media_type: str | None = None | |
| storage_reference: str | None = None | |
| class SubmittedCMRRequest(DatasetModel): | |
| """Metadata for the submitted CMR collection request.""" | |
| submitted_url: str = Field(min_length=1) | |
| submitted_native_id: str = Field(min_length=1) | |
| class DatasetIdentity(DatasetModel): | |
| """Authoritative CMR collection identity.""" | |
| concept_id: str = Field(min_length=1) | |
| native_id: str = Field(min_length=1) | |
| short_name: str = Field(min_length=1) | |
| version: str = Field(min_length=1) | |
| cmr_revision_id: int = Field(ge=1) | |
| legacy_entry_id: str | None = Field(default=None, exclude=True) | |
| def native_id_matches_source_fields(self) -> DatasetIdentity: | |
| """Require the representation-specific CMR Native ID derivation.""" | |
| expected = f"{self.short_name}_{self.version}" | |
| if self.native_id != expected: | |
| raise ValueError("native_id must equal short_name + '_' + version") | |
| return self | |
| class CMRSourceRecord(DatasetModel): | |
| """Preserved CMR source artifact and selected collection identity.""" | |
| request: SubmittedCMRRequest | |
| source_artifact: SourceArtifactReference | |
| identity: DatasetIdentity | |
| retrieved_at: str | None = None | |
| final_url: str | None = None | |
| class SealedExpertKeywordInput(DatasetModel): | |
| """Opaque reference to expert keywords excluded from blind inference.""" | |
| sealed_artifact_id: str = Field(min_length=1) | |
| sealed_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| keyword_count: int = Field(ge=0) | |
| class BlindCollectionView(DatasetModel): | |
| """Allow-listed dataset metadata safe for blind stages.""" | |
| identity: DatasetIdentity | |
| derived_native_id: str = Field(min_length=1) | |
| entry_title: str | None = None | |
| summary: str | None = None | |
| related_urls: tuple[dict[str, Any], ...] = Field(default_factory=tuple) | |
| blind_view_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| expert_keywords_excluded: Literal[True] = True | |
| def derived_identity_is_consistent(self) -> BlindCollectionView: | |
| if self.derived_native_id != self.identity.native_id: | |
| raise ValueError("derived_native_id must equal the validated Native ID") | |
| return self | |
| class READMECandidateState(str, Enum): # noqa: UP042 | |
| """Cardinality of usable exact READ-ME candidates.""" | |
| ZERO = "zero" | |
| ONE = "one" | |
| MULTIPLE = "multiple" | |
| class READMECandidate(DatasetModel): | |
| """One exact READ-ME entry discovered from source metadata.""" | |
| candidate_id: str = Field(min_length=1) | |
| source_index: int = Field(ge=0) | |
| url: str = Field(min_length=1) | |
| description: str | None = None | |
| type: str | None = None | |
| subtype: Literal["READ-ME"] = "READ-ME" | |
| source_entry: dict[str, Any] | |
| class READMEDiscoveryResult(DatasetModel): | |
| """Exact candidates discovered from one validated blind collection.""" | |
| identity: DatasetIdentity | |
| cmr_source_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| candidate_state: READMECandidateState | |
| candidates: tuple[READMECandidate, ...] = Field(default_factory=tuple) | |
| warnings: tuple[DatasetWarning, ...] = Field(default_factory=tuple) | |
| def cardinality_matches_state(self) -> READMEDiscoveryResult: | |
| expected = ( | |
| READMECandidateState.ZERO | |
| if not self.candidates | |
| else READMECandidateState.ONE | |
| if len(self.candidates) == 1 | |
| else READMECandidateState.MULTIPLE | |
| ) | |
| if self.candidate_state is not expected: | |
| raise ValueError("candidate_state must match candidate cardinality") | |
| return self | |
| class READMESelection(DatasetModel): | |
| """Explicit user selection of one discovered README candidate.""" | |
| candidate_id: str = Field(min_length=1) | |
| selected_url: str = Field(min_length=1) | |
| selected_at: str | None = None | |
| class ValidatedREADMESelection(DatasetModel): | |
| """Explicit selection bound to exact CMR discovery provenance.""" | |
| concept_id: str = Field(min_length=1) | |
| native_id: str = Field(min_length=1) | |
| short_name: str = Field(min_length=1) | |
| version: str = Field(min_length=1) | |
| cmr_revision_id: int = Field(ge=1) | |
| cmr_source_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| candidate_id: str = Field(min_length=1) | |
| source_index: int = Field(ge=0) | |
| selected_url: str = Field(min_length=1) | |
| source_entry: dict[str, Any] | |
| selected_at: str | |
| selection_binding_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| explicitly_confirmed: Literal[True] = True | |
| class RedirectRecord(DatasetModel): | |
| """One redirect retained as retrieval provenance.""" | |
| status_code: int = Field(ge=300, le=399) | |
| source_url: str = Field(min_length=1) | |
| target_url: str = Field(min_length=1) | |
| class READMERetrievalRecord(DatasetModel): | |
| """Sanitized provenance for a safely retrieved public README PDF.""" | |
| identity: DatasetIdentity | |
| cmr_source_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| selected_candidate_id: str = Field(min_length=1) | |
| selected_source_index: int = Field(ge=0) | |
| selected_source_entry: dict[str, Any] | |
| submitted_url: str = Field(min_length=1) | |
| final_url: str = Field(min_length=1) | |
| redirects: tuple[RedirectRecord, ...] = Field(default_factory=tuple) | |
| artifact: SourceArtifactReference | |
| retrieved_at: str | |
| status_code: int = Field(ge=200, le=299) | |
| response_metadata: dict[str, str] = Field(default_factory=dict) | |
| server_filename: str | None = None | |
| detected_media_type: Literal["application/pdf"] = "application/pdf" | |
| public_access_validated: Literal[True] = True | |
| public_address_validation: str = Field(min_length=1) | |
| document_validated: Literal[True] = True | |
| document_validation_method: str = Field(min_length=1) | |
| class ExtractedSeparator(DatasetModel): | |
| """Explicit separator omitted between adjacent blocks, if any.""" | |
| page_number: int = Field(ge=1) | |
| character_start: int = Field(ge=0) | |
| character_end: int = Field(ge=0) | |
| value: str | |
| separator_type: str = Field(min_length=1) | |
| policy_version: str = Field(min_length=1) | |
| class ExtractionLineage(DatasetModel): | |
| """Blind-safe lineage from a validated PDF to its CMR selection.""" | |
| identity: DatasetIdentity | |
| cmr_source_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| selected_candidate_id: str = Field(min_length=1) | |
| selected_source_index: int = Field(ge=0) | |
| selected_source_entry: dict[str, Any] | |
| selected_readme_url: str = Field(min_length=1) | |
| final_readme_url: str = Field(min_length=1) | |
| retrieval_timestamp: str = Field(min_length=1) | |
| retrieval_status_code: int = Field(ge=200, le=299) | |
| public_address_validation: str = Field(min_length=1) | |
| document_validation_method: str = Field(min_length=1) | |
| class ExtractedSection(DatasetModel): | |
| """Literal or derived section metadata with ordered membership.""" | |
| section_id: str = Field(pattern=r"^s[0-9]{4}$") | |
| source_order: int = Field(ge=0) | |
| heading: str | None = None | |
| normalized_heading: str | None = None | |
| heading_kind: HeadingKind | |
| label_source: SectionLabelSource | |
| heading_page_number: int | None = Field(default=None, ge=1) | |
| heading_character_start: int | None = Field(default=None, ge=0) | |
| heading_character_end: int | None = Field(default=None, ge=0) | |
| page_start: int = Field(ge=1) | |
| page_end: int = Field(ge=1) | |
| block_ids: tuple[str, ...] = Field(default_factory=tuple) | |
| def validate_section_provenance(self) -> ExtractedSection: | |
| if self.page_end < self.page_start: | |
| raise ValueError("page_end must be greater than or equal to page_start") | |
| spans = (self.heading_page_number, self.heading_character_start, self.heading_character_end) | |
| if self.label_source is SectionLabelSource.LITERAL: | |
| if self.heading is None or any(value is None for value in spans): | |
| raise ValueError("literal headings require exact page-text provenance") | |
| elif any(value is not None for value in spans[1:]): | |
| raise ValueError("derived headings cannot claim literal character spans") | |
| return self | |
| class ExtractedBlock(DatasetModel): | |
| """Exact page-text slice with complete source and policy provenance.""" | |
| block_id: str = Field(pattern=r"^p[0-9]{4}-b[0-9]{4}$") | |
| document_id: str = Field(min_length=1) | |
| pdf_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| page_id: str = Field(min_length=1) | |
| page_start: int = Field(ge=1) | |
| page_end: int = Field(ge=1) | |
| section_id: str = Field(pattern=r"^s[0-9]{4}$") | |
| section_heading: str | None = None | |
| section_label_source: SectionLabelSource | |
| text: str | |
| character_start: int = Field(ge=0) | |
| character_end: int = Field(ge=0) | |
| source_order: int = Field(ge=0) | |
| extraction_version: str = Field(min_length=1) | |
| splitting_policy_version: str = Field(min_length=1) | |
| extraction_warnings: tuple[DatasetWarning, ...] = Field(default_factory=tuple) | |
| def offsets_and_pages_are_ordered(self) -> ExtractedBlock: | |
| if self.page_end < self.page_start: | |
| raise ValueError("page_end must be greater than or equal to page_start") | |
| if self.character_end < self.character_start: | |
| raise ValueError("character_end must be greater than or equal to character_start") | |
| if len(self.text) != self.character_end - self.character_start: | |
| raise ValueError("block text length must match its page-text character span") | |
| return self | |
| class ExtractedPage(DatasetModel): | |
| """One-based deterministic page extraction result.""" | |
| page_id: str = Field(min_length=1) | |
| page_number: int = Field(ge=1) | |
| source_order: int = Field(ge=0) | |
| text: str | |
| text_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| character_count: int = Field(ge=0) | |
| extraction_status: PageExtractionStatus | |
| width_points: float | None = Field(default=None, gt=0) | |
| height_points: float | None = Field(default=None, gt=0) | |
| coordinate_system: str | None = None | |
| block_ids: tuple[str, ...] = Field(default_factory=tuple) | |
| warnings: tuple[DatasetWarning, ...] = Field(default_factory=tuple) | |
| def validate_page_text_metrics(self) -> ExtractedPage: | |
| if self.character_count != len(self.text): | |
| raise ValueError("page character_count must equal stored text length") | |
| return self | |
| class DatasetExtractionManifest(DatasetModel): | |
| """Formal deterministic PDF extraction and provenance output.""" | |
| schema_version: str = Field(min_length=1) | |
| document: SourceArtifactReference | |
| lineage: ExtractionLineage | |
| document_id: str = Field(pattern=r"^sha256:[0-9a-f]{64}$") | |
| extraction_method: str = Field(min_length=1) | |
| extraction_version: str = Field(min_length=1) | |
| page_text_policy_version: str = Field(min_length=1) | |
| heading_policy_version: str = Field(min_length=1) | |
| section_policy_version: str = Field(min_length=1) | |
| segmentation_version: str = Field(min_length=1) | |
| identifier_policy_version: str = Field(min_length=1) | |
| separator_policy_version: str = Field(min_length=1) | |
| page_numbering: Literal["one_based"] = "one_based" | |
| usable_for_classification: Literal[True] = True | |
| page_count: int = Field(ge=1) | |
| character_count: int = Field(ge=1) | |
| block_count: int = Field(ge=1) | |
| configured_limits: dict[str, int] | |
| pages: tuple[ExtractedPage, ...] | |
| sections: tuple[ExtractedSection, ...] | |
| blocks: tuple[ExtractedBlock, ...] | |
| excluded_separators: tuple[ExtractedSeparator, ...] = Field(default_factory=tuple) | |
| warnings: tuple[DatasetWarning, ...] = Field(default_factory=tuple) | |
| def validate_manifest_invariants(self) -> DatasetExtractionManifest: | |
| if self.document_id != f"sha256:{self.document.sha256}": | |
| raise ValueError("document_id must derive from the exact PDF SHA-256") | |
| if self.page_count != len(self.pages) or self.block_count != len(self.blocks): | |
| raise ValueError("manifest counts must match contained records") | |
| if self.character_count != sum(page.character_count for page in self.pages): | |
| raise ValueError("manifest character_count must match pages") | |
| if [page.page_number for page in self.pages] != list(range(1, len(self.pages) + 1)): | |
| raise ValueError("pages must be consecutive and one-based") | |
| if [page.source_order for page in self.pages] != list(range(len(self.pages))): | |
| raise ValueError("page source_order values must be consecutive") | |
| if [block.source_order for block in self.blocks] != list(range(len(self.blocks))): | |
| raise ValueError("block source_order values must be consecutive") | |
| page_by_number = {page.page_number: page for page in self.pages} | |
| for block in self.blocks: | |
| page = page_by_number[block.page_start] | |
| if block.page_start != block.page_end or page.page_id != block.page_id: | |
| raise ValueError("blocks must be page-bounded and reference the correct page") | |
| if page.text[block.character_start : block.character_end] != block.text: | |
| raise ValueError("block text must be an exact page-text slice") | |
| for page in self.pages: | |
| page_blocks = [block for block in self.blocks if block.page_start == page.page_number] | |
| if "".join(block.text for block in page_blocks) != page.text: | |
| raise ValueError("ordered blocks must exactly reconstruct approved page text") | |
| return self | |
| class ExtractionPreclassificationFailure(DatasetModel): | |
| """Fatal extraction result with an explicit zero-inference boundary.""" | |
| error: DatasetErrorRecord | |
| classification_attempted: Literal[False] = False | |
| model_calls: Literal[0] = 0 | |
| usable_for_classification: Literal[False] = False | |
| manifest: Literal[None] = None | |
| comparison: Literal[None] = None | |
| class ConfidenceSignal(DatasetModel): | |
| """Explicitly uncalibrated confidence signal.""" | |
| value: float = Field(ge=0.0, le=1.0) | |
| calibrated: Literal[False] = False | |
| class ProductScopeDecision(DatasetModel): | |
| """Product applicability decision for one source block.""" | |
| scope_type: ProductScopeType | |
| applicable_short_names: tuple[str, ...] = Field(default_factory=tuple) | |
| family_label: str | None = None | |
| product_match_confidence: ConfidenceSignal | |
| rationale: str = Field(min_length=1) | |
| qualifying_block_ids: tuple[str, ...] = Field(default_factory=tuple) | |
| review_required: bool = False | |
| class ScientificRelevance(str, Enum): # noqa: UP042 | |
| """Scientific relevance assigned without making GCMD recommendations.""" | |
| RELEVANT = "relevant" | |
| NOT_RELEVANT = "not_relevant" | |
| CONTEXT_ONLY = "context_only" | |
| class GroupMembership(str, Enum): # noqa: UP042 | |
| """Whether a grouped block requires a decision or supplies context only.""" | |
| DECISION_TARGET = "decision_target" | |
| CONTEXT = "context" | |
| class PacketReadiness(str, Enum): # noqa: UP042 | |
| """Whether a packet may cross the later classifier boundary.""" | |
| CLASSIFICATION_READY = "classification_ready" | |
| REVIEW_REQUIRED = "review_required" | |
| class EvidenceSelectionDecisionV1(DatasetModel): | |
| """Historical v1 response retained for persisted-artifact recovery only.""" | |
| block_id: str = Field(min_length=1) | |
| selected: bool | |
| scientific_relevance: ScientificRelevance | |
| evidence_role: EvidenceRole | |
| scope_type: ProductScopeType | |
| evidence_eligibility: EvidenceEligibility | |
| product_match_confidence: ConfidenceSignal | |
| qualifying_block_ids: tuple[str, ...] = Field(default_factory=tuple) | |
| qualifying_declaration_ids: tuple[str, ...] = Field(default_factory=tuple) | |
| rationale: str = Field(min_length=1) | |
| review_required: bool | |
| prompt_version: Literal["dataset_evidence_selection_v1"] = "dataset_evidence_selection_v1" | |
| warnings: tuple[str, ...] = Field(default_factory=tuple) | |
| class EvidenceSelectionDecision(DatasetModel): | |
| """Strict v2 model decision for one supplied decision-target block.""" | |
| block_id: str = Field(min_length=1) | |
| selected: bool | |
| scientific_relevance: ScientificRelevance | |
| evidence_role: EvidenceRole | |
| scope_type: ProductScopeType | |
| evidence_eligibility: EvidenceEligibility | |
| product_match_confidence: ConfidenceSignal | |
| qualifying_block_ids: tuple[str, ...] = Field(default_factory=tuple) | |
| qualifying_declaration_ids: tuple[str, ...] = Field(default_factory=tuple) | |
| rationale: str = Field(min_length=1) | |
| review_required: bool | |
| prompt_version: Literal["dataset_evidence_selection_v2"] = "dataset_evidence_selection_v2" | |
| warnings: tuple[str, ...] = Field(default_factory=tuple) | |
| class EvidenceSelectionModelResponse(DatasetModel): | |
| """Strict response for one bounded evidence-selection group.""" | |
| decisions: tuple[EvidenceSelectionDecision, ...] | |
| class EvidenceMinimizationDecision(DatasetModel): | |
| """One source-ID-only decision in the bounded packet-minimality pass.""" | |
| block_id: str = Field(min_length=1) | |
| retain: bool | |
| removal_reason: ( | |
| Literal["exact_duplicate", "redundant_science_evidence", "non_independent_science"] | None | |
| ) = None | |
| prompt_version: Literal["dataset_evidence_minimization_v1"] = "dataset_evidence_minimization_v1" | |
| class EvidenceMinimizationModelResponse(DatasetModel): | |
| """Strict response that can only retain or remove supplied selected IDs.""" | |
| decisions: tuple[EvidenceMinimizationDecision, ...] | |
| class EvidenceExclusionReason(str, Enum): # noqa: UP042 | |
| """Stable, content-free audit reasons for candidate or packet exclusion.""" | |
| PRODUCT_REVIEW_REQUIRED = "product_review_required" | |
| CONFIDENCE_BELOW_ACCEPTANCE = "confidence_below_acceptance" | |
| APPLICABILITY_UNTRACEABLE = "applicability_untraceable" | |
| DETERMINISTIC_NON_SCIENCE = "deterministic_non_science" | |
| MODEL_NOT_SELECTED = "model_not_selected" | |
| NON_INDEPENDENT_ROLE = "non_independent_role" | |
| DUPLICATE_SOURCE = "duplicate_source" | |
| MINIMALITY_REMOVED = "minimality_removed" | |
| class EvidenceExclusionRecord(DatasetModel): | |
| """Safe audit record containing identity and policy reason, never source text.""" | |
| block_id: str = Field(min_length=1) | |
| reason: EvidenceExclusionReason | |
| phase: Literal["candidate", "selection", "consolidation", "minimality"] | |
| class EvidencePacketBlock(DatasetModel): | |
| """Authoritative source block plus clearly separate derived annotations.""" | |
| block: ExtractedBlock | |
| block_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| evidence_role: EvidenceRole | |
| scientific_relevance: ScientificRelevance = ScientificRelevance.RELEVANT | |
| product_scope: ProductScopeDecision | |
| evidence_eligibility: EvidenceEligibility = EvidenceEligibility.ELIGIBLE | |
| selection_rationale: str = Field(default="Selected scientific source evidence.", min_length=1) | |
| qualifying_block_ids: tuple[str, ...] = Field(default_factory=tuple) | |
| qualifying_declaration_ids: tuple[str, ...] = Field(default_factory=tuple) | |
| context_for_block_ids: tuple[str, ...] = Field(default_factory=tuple) | |
| review_required: bool = False | |
| class PacketComponentHashes(DatasetModel): | |
| """Stable component identities for the logical evidence packet.""" | |
| target_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| source_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| ordered_block_identity_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| ordered_source_text_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| product_scope_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| selection_decisions_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| policy_versions_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| limit_lineage_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| class PacketLimitRecord(DatasetModel): | |
| """Versioned packet measurement and bounded reduction lineage.""" | |
| policy_version: str = Field(min_length=1) | |
| measurement_version: str = Field(min_length=1) | |
| max_blocks: int = Field(gt=0) | |
| max_characters: int = Field(gt=0) | |
| preferred_blocks: int | None = Field(default=None, gt=0) | |
| preferred_characters: int | None = Field(default=None, gt=0) | |
| initial_block_ids: tuple[str, ...] | |
| final_block_ids: tuple[str, ...] | |
| removed_block_ids: tuple[str, ...] = Field(default_factory=tuple) | |
| removal_rationales: dict[str, str] = Field(default_factory=dict) | |
| initial_block_count: int = Field(ge=0) | |
| final_block_count: int = Field(ge=0) | |
| initial_character_count: int = Field(ge=0) | |
| final_character_count: int = Field(ge=0) | |
| initial_estimated_tokens: int = Field(ge=0) | |
| final_estimated_tokens: int = Field(ge=0) | |
| reduction_required: bool | |
| reduction_passes: int = Field(ge=0, le=1) | |
| minimality_policy_version: str | None = None | |
| minimality_model_calls: int = Field(default=0, ge=0) | |
| within_limits: bool | |
| class DatasetEvidencePacket(DatasetModel): | |
| """Formal immutable and source-faithful later-classifier input.""" | |
| schema_version: str = Field(min_length=1) | |
| packet_id: str = Field(min_length=1) | |
| packet_version: str = Field(min_length=1) | |
| target: DatasetIdentity | |
| cmr_source_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| readme_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| extraction_manifest_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| coverage_mode: CoverageMode | |
| coverage_confidence: ConfidenceSignal | |
| selection_version: str = Field(min_length=1) | |
| grouping_policy_version: str = Field(default="dataset_evidence_grouping_v2", min_length=1) | |
| evidence_role_policy_version: str = Field(default="dataset_evidence_roles_v2", min_length=1) | |
| minimality_policy_version: str = Field(default="dataset_evidence_minimization_v1", min_length=1) | |
| response_schema_version: str = Field( | |
| default="dataset_evidence_selection_schema_v2", min_length=1 | |
| ) | |
| blocks: tuple[EvidencePacketBlock, ...] | |
| excluded_block_ids: tuple[str, ...] = Field(default_factory=tuple) | |
| ambiguous_block_ids: tuple[str, ...] = Field(default_factory=tuple) | |
| context_only_block_ids: tuple[str, ...] = Field(default_factory=tuple) | |
| character_count: int = Field(ge=0) | |
| estimated_tokens: int | None = Field(default=None, ge=0) | |
| packet_readiness: PacketReadiness = PacketReadiness.CLASSIFICATION_READY | |
| review_status: DatasetReviewStatus = DatasetReviewStatus.NOT_REQUIRED | |
| limit_record: PacketLimitRecord | None = None | |
| component_hashes: PacketComponentHashes | None = None | |
| warnings: tuple[DatasetWarning, ...] = Field(default_factory=tuple) | |
| packet_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| def validate_packet_shape(self) -> DatasetEvidencePacket: | |
| block_ids = [item.block.block_id for item in self.blocks] | |
| if len(block_ids) != len(set(block_ids)): | |
| raise ValueError("evidence packet block IDs must be unique") | |
| if block_ids and [item.block.source_order for item in self.blocks] != sorted( | |
| item.block.source_order for item in self.blocks | |
| ): | |
| raise ValueError("evidence packet blocks must preserve source order") | |
| if self.character_count != sum(len(item.block.text) for item in self.blocks): | |
| raise ValueError("packet character_count must match exact block text") | |
| if self.blocks and not any( | |
| item.scientific_relevance is ScientificRelevance.RELEVANT for item in self.blocks | |
| ): | |
| raise ValueError("a packet cannot contain only context blocks") | |
| return self | |
| class DatasetDeterministicValidation(DatasetModel): | |
| """Vocabulary-integrity result for one dataset classification.""" | |
| valid: bool | |
| errors: tuple[DatasetErrorRecord, ...] = Field(default_factory=tuple) | |
| warnings: tuple[DatasetWarning, ...] = Field(default_factory=tuple) | |
| class KeywordEvidenceConfidence(DatasetModel): | |
| """Uncalibrated hierarchy-stage confidence signals.""" | |
| topic: float | None = Field(default=None, ge=0.0, le=1.0) | |
| term: float | None = Field(default=None, ge=0.0, le=1.0) | |
| variable_level_1: float | None = Field(default=None, ge=0.0, le=1.0) | |
| variable_level_2: float | None = Field(default=None, ge=0.0, le=1.0) | |
| variable_level_3: float | None = Field(default=None, ge=0.0, le=1.0) | |
| final: float | None = Field(default=None, ge=0.0, le=1.0) | |
| calibrated: Literal[False] = False | |
| class DatasetSemanticValidatorState(str, Enum): # noqa: UP042 | |
| """Independent semantic-validator execution state for dataset classification.""" | |
| NOT_RUN_BY_POLICY = "not_run_by_policy" | |
| NOT_APPLICABLE = "not_applicable" | |
| class DatasetEvidenceCitation(DatasetModel): | |
| """Validated reference back to one immutable evidence-packet block.""" | |
| block_id: str = Field(min_length=1) | |
| page_start: int = Field(ge=1) | |
| page_end: int = Field(ge=1) | |
| section_id: str = Field(min_length=1) | |
| section_heading: str | None = None | |
| evidence_role: EvidenceRole | |
| product_scope_type: ProductScopeType | |
| block_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| independently_eligible: bool | |
| class DatasetClassification(DatasetModel): | |
| """One blind, authoritative GCMD classification for a dataset.""" | |
| UUID: str = Field(min_length=1) | |
| name: str = Field(min_length=1) | |
| level: HierarchyLevel | |
| canonical_path: str = Field(min_length=1) | |
| path_components: tuple[str, ...] = Field(min_length=1) | |
| topic: str = Field(min_length=1) | |
| term: str | None = None | |
| cited_block_ids: tuple[str, ...] = Field(min_length=1) | |
| citations: tuple[DatasetEvidenceCitation, ...] = Field(min_length=1) | |
| deepest_supported_uuid: str = Field(min_length=1) | |
| decision_prompt_versions: tuple[str, ...] = Field(default_factory=tuple) | |
| product_match_confidence: ConfidenceSignal | |
| keyword_evidence_confidence: KeywordEvidenceConfidence | |
| reason_for_stopping: str = Field(min_length=1) | |
| deterministic_validation: DatasetDeterministicValidation | |
| final_status: DatasetClassificationFinalStatus | |
| review_required: bool = False | |
| def accepted_results_are_valid(self) -> DatasetClassification: | |
| if ( | |
| self.final_status | |
| in { | |
| DatasetClassificationFinalStatus.ACCEPTED, | |
| DatasetClassificationFinalStatus.REDUCED_TO_ANCESTOR, | |
| } | |
| and not self.deterministic_validation.valid | |
| ): | |
| raise ValueError("accepted or reduced classifications require valid vocabulary data") | |
| return self | |
| class BlindResultSeal(DatasetModel): | |
| """Hash and persistence metadata for a completed blind result.""" | |
| blind_result_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| evidence_packet_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| persisted_at: str = Field(min_length=1) | |
| persistence_reference: str = Field(min_length=1) | |
| expert_keywords_excluded: Literal[True] = True | |
| class DatasetProcessingMetadata(DatasetModel): | |
| """Reproducibility metadata for dataset processing.""" | |
| run_id: str | None = None | |
| started_at: str | None = None | |
| completed_at: str | None = None | |
| application_version: str | None = None | |
| vocabulary_hash: str | None = None | |
| configuration_hash: str | None = None | |
| prompt_versions: dict[str, str] = Field(default_factory=dict) | |
| model_provider: str | None = None | |
| model_name: str | None = None | |
| model_calls: int = Field(default=0, ge=0) | |
| input_tokens: int | None = Field(default=None, ge=0) | |
| output_tokens: int | None = Field(default=None, ge=0) | |
| estimated_cost: float | None = Field(default=None, ge=0.0) | |
| cache_used: bool = False | |
| class DatasetClassificationResult(DatasetModel): | |
| """Top-level dataset classification result contract.""" | |
| schema_version: str = Field(min_length=1) | |
| dataset_key: str = Field(min_length=1) | |
| identity: DatasetIdentity | |
| cmr_source: CMRSourceRecord | None = None | |
| sealed_expert_keywords: SealedExpertKeywordInput | None = None | |
| readme_candidates: tuple[READMECandidate, ...] = Field(default_factory=tuple) | |
| selected_readme: READMESelection | None = None | |
| readme_artifact: SourceArtifactReference | None = None | |
| extraction_manifest_reference: str | None = None | |
| product_resolution_reference: str | None = None | |
| evidence_packet_reference: str | None = None | |
| evidence_packet_sha256: str | None = Field(default=None, pattern=r"^[0-9a-f]{64}$") | |
| coverage_mode: CoverageMode | None = None | |
| semantic_validator_state: DatasetSemanticValidatorState = ( | |
| DatasetSemanticValidatorState.NOT_APPLICABLE | |
| ) | |
| no_classification_cited_block_ids: tuple[str, ...] = Field(default_factory=tuple) | |
| classification_attempted: bool | |
| processing_status: DatasetProcessingStatus | |
| classification_outcome: DatasetClassificationOutcome | None = None | |
| classifications: tuple[DatasetClassification, ...] = Field(default_factory=tuple) | |
| no_classification_reason: str | None = None | |
| blind_result: BlindResultSeal | None = None | |
| comparison_reference: str | None = None | |
| review_status: DatasetReviewStatus = DatasetReviewStatus.NOT_REQUIRED | |
| warnings: tuple[DatasetWarning, ...] = Field(default_factory=tuple) | |
| errors: tuple[DatasetErrorRecord, ...] = Field(default_factory=tuple) | |
| hierarchy_attempt_diagnostics: tuple[DatasetHierarchyAttemptDiagnostic, ...] = Field( | |
| default_factory=tuple | |
| ) | |
| processing_metadata: DatasetProcessingMetadata = Field( | |
| default_factory=DatasetProcessingMetadata | |
| ) | |
| def validate_result_scope(self) -> DatasetClassificationResult: | |
| if self.dataset_key != self.identity.concept_id: | |
| raise ValueError("dataset_key must equal identity.concept_id") | |
| if self.cmr_source is not None and self.cmr_source.identity != self.identity: | |
| raise ValueError("cmr_source identity must match result identity") | |
| if not self.classification_attempted: | |
| if self.classification_outcome is not None: | |
| raise ValueError("unattempted classification cannot have an outcome") | |
| if self.classifications: | |
| raise ValueError("unattempted classification cannot contain classifications") | |
| if self.blind_result is not None: | |
| raise ValueError("unattempted classification cannot have a blind result") | |
| if self.comparison_reference is not None: | |
| raise ValueError("unattempted classification cannot have comparison output") | |
| if self.processing_metadata.model_calls != 0: | |
| raise ValueError("unattempted classification requires zero model calls") | |
| else: | |
| if self.coverage_mode is None: | |
| raise ValueError("attempted dataset processing requires coverage identity") | |
| semantic_no_science = ( | |
| self.evidence_packet_sha256 is None | |
| and self.classification_outcome is DatasetClassificationOutcome.NOT_CLASSIFIED | |
| and not self.classifications | |
| and bool(self.no_classification_reason) | |
| ) | |
| if semantic_no_science: | |
| if ( | |
| self.semantic_validator_state | |
| is not DatasetSemanticValidatorState.NOT_APPLICABLE | |
| ): | |
| raise ValueError( | |
| "pre-classifier semantic stops require validator not applicable" | |
| ) | |
| elif ( | |
| self.evidence_packet_sha256 is None | |
| or self.semantic_validator_state | |
| is not DatasetSemanticValidatorState.NOT_RUN_BY_POLICY | |
| ): | |
| raise ValueError("GCMD classification requires a packet and validator policy state") | |
| if ( | |
| self.classification_attempted | |
| and self.processing_status is DatasetProcessingStatus.COMPLETED | |
| ): | |
| if self.classification_outcome is None: | |
| raise ValueError("completed attempted classification requires an outcome") | |
| if ( | |
| self.classification_outcome is DatasetClassificationOutcome.CLASSIFIED | |
| and not self.classifications | |
| ): | |
| raise ValueError("classified dataset results require classifications") | |
| if self.classification_outcome is DatasetClassificationOutcome.NOT_CLASSIFIED and ( | |
| self.classifications or not self.no_classification_reason | |
| ): | |
| raise ValueError("not_classified requires no classifications and a reason") | |
| if self.comparison_reference is not None and self.blind_result is None: | |
| raise ValueError("comparison output requires a persisted blind result") | |
| return self | |
| class ResolvedExpertKeyword(DatasetModel): | |
| """Expert keyword deterministically resolved to local vocabulary.""" | |
| source_index: int = Field(ge=0) | |
| UUID: str = Field(min_length=1) | |
| name: str = Field(min_length=1) | |
| canonical_path: str = Field(min_length=1) | |
| class UnresolvedExpertKeyword(DatasetModel): | |
| """Expert source entry that could not be resolved unambiguously.""" | |
| source_index: int = Field(ge=0) | |
| source_entry: dict[str, Any] | |
| reason: str = Field(min_length=1) | |
| class DatasetComparisonRecord(DatasetModel): | |
| """One additive post-inference comparison decision.""" | |
| blind_classification_uuid: str = Field(min_length=1) | |
| expert_keyword_uuids: tuple[str, ...] = Field(default_factory=tuple) | |
| hierarchy_relationship: str = Field(min_length=1) | |
| category: ComparisonCategory | |
| rationale: str = Field(min_length=1) | |
| review_required: bool | |
| class DatasetComparison(DatasetModel): | |
| """Formal post-inference comparison contract.""" | |
| schema_version: str = Field(min_length=1) | |
| dataset_key: str = Field(min_length=1) | |
| blind_result_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| sealed_expert_keyword_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| vocabulary_hash: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| resolved_expert_keywords: tuple[ResolvedExpertKeyword, ...] = Field(default_factory=tuple) | |
| unresolved_expert_keywords: tuple[UnresolvedExpertKeyword, ...] = Field(default_factory=tuple) | |
| comparisons: tuple[DatasetComparisonRecord, ...] = Field(default_factory=tuple) | |
| review_status: DatasetReviewStatus | |
| warnings: tuple[DatasetWarning, ...] = Field(default_factory=tuple) | |
| errors: tuple[DatasetErrorRecord, ...] = Field(default_factory=tuple) | |
| class CMRRedirectProvenance(DatasetModel): | |
| """One manually validated CMR redirect.""" | |
| status_code: int = Field(ge=300, le=399) | |
| source_url: str = Field(min_length=1) | |
| target_url: str = Field(min_length=1) | |
| class CMRRetrievedSource(DatasetModel): | |
| """Exact CMR response bytes plus derived parsed representation.""" | |
| submitted_url: str = Field(min_length=1) | |
| final_url: str = Field(min_length=1) | |
| redirect_history: tuple[CMRRedirectProvenance, ...] = Field(default_factory=tuple) | |
| retrieved_at: str = Field(min_length=1) | |
| status_code: int = Field(ge=100, le=599) | |
| response_metadata: dict[str, str] = Field(default_factory=dict) | |
| response_bytes: bytes | |
| source_text: str | |
| sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| parsed_json: dict[str, Any] | |
| class SealedExpertKeywordPayload(DatasetModel): | |
| """Original expert assignments held outside all blind-stage interfaces.""" | |
| dataset_key: str = Field(min_length=1) | |
| source_path: Literal["umm.ScienceKeywords"] = "umm.ScienceKeywords" | |
| science_keywords: tuple[dict[str, Any], ...] = Field(default_factory=tuple) | |
| payload_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| class ResolvedCMRCollection(DatasetModel): | |
| """Validated source, authoritative identity, blind view, and sealed payload.""" | |
| dataset_key: str = Field(min_length=1) | |
| submitted_native_id: str = Field(min_length=1) | |
| returned_native_id: str = Field(min_length=1) | |
| derived_native_id: str = Field(min_length=1) | |
| identity: DatasetIdentity | |
| source: CMRRetrievedSource | |
| source_item: dict[str, Any] | |
| blind_view: BlindCollectionView | |
| sealed_expert_keywords: SealedExpertKeywordPayload | |
| def identifiers_are_consistent(self) -> ResolvedCMRCollection: | |
| values = { | |
| self.submitted_native_id, | |
| self.returned_native_id, | |
| self.derived_native_id, | |
| self.identity.native_id, | |
| } | |
| if len(values) != 1: | |
| raise ValueError("submitted, returned, derived, and typed Native IDs must match") | |
| if self.dataset_key != self.identity.concept_id: | |
| raise ValueError("dataset_key must equal the CMR Concept ID") | |
| return self | |
| class CMRPreclassificationFailure(DatasetModel): | |
| """Explicit CMR-stage failure that cannot be mistaken for classification.""" | |
| error: DatasetErrorRecord | |
| classification_attempted: Literal[False] = False | |
| model_calls: Literal[0] = 0 | |
| classifications: tuple[()] = () | |
| classification_outcome: Literal[None] = None | |
| comparison: Literal[None] = None | |
| class READMEPreclassificationOutcome(DatasetModel): | |
| """Discovery/retrieval outcome that explicitly performed no inference.""" | |
| reason_code: str = Field(min_length=1) | |
| processing_status: DatasetProcessingStatus | |
| classification_attempted: Literal[False] = False | |
| model_calls: Literal[0] = 0 | |
| classifications: tuple[()] = () | |
| classification_outcome: Literal[None] = None | |
| extraction: Literal[None] = None | |
| comparison: Literal[None] = None | |
| class ProductSignalType(str, Enum): # noqa: UP042 | |
| """Versioned deterministic product-applicability signal kinds.""" | |
| EXACT_SHORT_NAME = "exact_short_name" | |
| EXACT_NATIVE_ID = "exact_native_id" | |
| EXACT_VERSION = "exact_version" | |
| EXPLICIT_ALIAS_DECLARATION = "explicit_alias_declaration" | |
| ENUMERATED_PRODUCT_LIST = "enumerated_product_list" | |
| TARGET_IN_ENUMERATION = "target_in_enumeration" | |
| OTHER_PRODUCT_IN_ENUMERATION = "other_product_in_enumeration" | |
| FAMILY_DECLARATION = "family_declaration" | |
| TARGET_FAMILY_MEMBERSHIP = "target_family_membership" | |
| UNPROVEN_FAMILY_MEMBERSHIP = "unproven_family_membership" | |
| TEMPLATE_HEADING = "template_heading" | |
| DOCUMENT_WIDE_CONTEXT = "document_wide_context" | |
| SHARED_CONTEXT = "shared_context" | |
| OTHER_PRODUCT_RESTRICTION = "other_product_restriction" | |
| UNRESOLVED_SCOPE = "unresolved_scope" | |
| NORMALIZED_TARGET_ALIAS = "normalized_target_alias" | |
| OTHER_PRODUCT_HEADING = "other_product_heading" | |
| class DeclarationKind(str, Enum): # noqa: UP042 | |
| """Deterministically recognized declaration structure.""" | |
| IDENTITY = "identity" | |
| ENUMERATED_PRODUCTS = "enumerated_products" | |
| PRODUCT_FAMILY = "product_family" | |
| DOCUMENT_WIDE = "document_wide" | |
| OTHER_PRODUCT_ONLY = "other_product_only" | |
| class AliasSource(str, Enum): # noqa: UP042 | |
| """Approved provenance for a product identity alias.""" | |
| VALIDATED_CMR_IDENTITY = "validated_cmr_identity" | |
| DETERMINISTIC_DERIVATION = "deterministic_derivation" | |
| DETERMINISTIC_NORMALIZATION = "deterministic_normalization" | |
| README_DECLARATION = "readme_declaration" | |
| class ProductReviewReason(str, Enum): # noqa: UP042 | |
| """Versioned reasons for block- or dataset-level product review.""" | |
| MODEL_REVIEW_ADVICE = "model_review_advice" | |
| TARGET_BLOCK_CONFIDENCE_BELOW_THRESHOLD = "target_block_confidence_below_threshold" | |
| AMBIGUOUS_PRODUCT_SCOPE = "ambiguous_product_scope" | |
| MIXED_PRODUCT_TRANSITION = "mixed_product_transition" | |
| DOCUMENT_TARGET_CONFIDENCE_BELOW_THRESHOLD = "document_target_confidence_below_threshold" | |
| VAGUE_OR_UNDETERMINED_COVERAGE = "vague_or_undetermined_coverage" | |
| TARGET_INCLUSION_UNPROVEN = "target_inclusion_unproven" | |
| FAMILY_MEMBERSHIP_UNPROVEN = "family_membership_unproven" | |
| DOCUMENT_CONTRADICTION = "document_contradiction" | |
| NO_DEFENSIBLE_TARGET_PASSAGE = "no_defensible_target_passage" | |
| class DeclarationLocality(DatasetModel): | |
| """Deterministic forward-only applicability range for one declaration.""" | |
| policy_version: str = Field(min_length=1) | |
| anchor_block_id: str = Field(min_length=1) | |
| section_id: str = Field(min_length=1) | |
| source_order_start: int = Field(ge=0) | |
| source_order_end: int = Field(ge=0) | |
| allowed_block_ids: tuple[str, ...] = Field(min_length=1) | |
| explicit_backward_scope: bool = False | |
| def validate_locality(self) -> DeclarationLocality: | |
| if self.source_order_end < self.source_order_start: | |
| raise ValueError("declaration locality source order must be ordered") | |
| if len(self.allowed_block_ids) != len(set(self.allowed_block_ids)): | |
| raise ValueError("declaration locality block IDs must be unique") | |
| if self.anchor_block_id not in self.allowed_block_ids: | |
| raise ValueError("declaration locality must contain its anchor block") | |
| return self | |
| class ResolutionDeterminationMethod(str, Enum): # noqa: UP042 | |
| """Whether a result came from rules, model analysis, or both.""" | |
| DETERMINISTIC = "deterministic" | |
| MODEL = "model" | |
| HYBRID = "hybrid" | |
| class DeterministicProductSignal(DatasetModel): | |
| """Exact source span and context for a deterministic product signal.""" | |
| signal_id: str = Field(min_length=1) | |
| signal_type: ProductSignalType | |
| block_id: str = Field(min_length=1) | |
| page_number: int = Field(ge=1) | |
| section_id: str = Field(min_length=1) | |
| literal_value: str | |
| normalized_value: str | None = None | |
| character_start: int = Field(ge=0) | |
| character_end: int = Field(ge=0) | |
| method_version: str = Field(min_length=1) | |
| def validate_signal_span(self) -> DeterministicProductSignal: | |
| if self.character_end < self.character_start: | |
| raise ValueError("signal character span must be ordered") | |
| if len(self.literal_value) != self.character_end - self.character_start: | |
| raise ValueError("literal signal length must match its block-text span") | |
| return self | |
| class ProductDeclaration(DatasetModel): | |
| """Auditable product or family declaration assembled from source signals.""" | |
| declaration_id: str = Field(min_length=1) | |
| kind: DeclarationKind | |
| block_id: str = Field(min_length=1) | |
| signal_ids: tuple[str, ...] = Field(min_length=1) | |
| product_values: tuple[str, ...] = Field(default_factory=tuple) | |
| family_label: str | None = None | |
| target_included: bool | |
| contradictory: bool = False | |
| method_version: str = Field(min_length=1) | |
| locality: DeclarationLocality | None = None | |
| class ProductAliasProvenance(DatasetModel): | |
| """Allowed product identity with explicit non-model provenance.""" | |
| alias: str = Field(min_length=1) | |
| source: AliasSource | |
| derivation_method: str = Field(min_length=1) | |
| method_version: str = Field(min_length=1) | |
| source_block_id: str | None = None | |
| character_start: int | None = Field(default=None, ge=0) | |
| character_end: int | None = Field(default=None, ge=0) | |
| source_literal: str | None = None | |
| normalization_rule: str | None = None | |
| class CoverageAnalysisInput(DatasetModel): | |
| """Minimum blind-stage input for coverage and passage resolution.""" | |
| target: DatasetIdentity | |
| derived_native_id: str = Field(min_length=1) | |
| entry_title: str | None = None | |
| summary: str | None = None | |
| extraction_manifest: DatasetExtractionManifest | |
| deterministic_analysis_version: Literal["dataset_product_signals_v3"] = ( | |
| "dataset_product_signals_v3" | |
| ) | |
| declaration_locality_policy_version: Literal["dataset_declaration_locality_v3"] = ( | |
| "dataset_declaration_locality_v3" | |
| ) | |
| downstream_eligibility_policy_version: Literal["dataset_downstream_eligibility_v5"] = ( | |
| "dataset_downstream_eligibility_v5" | |
| ) | |
| review_aggregation_policy_version: Literal["dataset_product_review_v5"] = ( | |
| "dataset_product_review_v5" | |
| ) | |
| mixed_product_transition_policy_version: Literal["dataset_mixed_product_transition_v1"] = ( | |
| "dataset_mixed_product_transition_v1" | |
| ) | |
| coverage_prompt_version: Literal["dataset_coverage_v3"] = "dataset_coverage_v3" | |
| product_resolution_prompt_version: Literal["dataset_product_resolution_v3"] = ( | |
| "dataset_product_resolution_v3" | |
| ) | |
| def validate_blind_lineage(self) -> CoverageAnalysisInput: | |
| if self.target != self.extraction_manifest.lineage.identity: | |
| raise ValueError("coverage target must match extraction lineage identity") | |
| if self.derived_native_id != self.target.native_id: | |
| raise ValueError("derived Native ID must match validated target identity") | |
| return self | |
| class CoverageModelResponseV2(DatasetModel): | |
| """Legacy v2 coverage response retained only for persisted-artifact recovery.""" | |
| coverage_mode: CoverageMode | |
| qualifying_block_ids: tuple[str, ...] = Field(default_factory=tuple) | |
| rejected_block_ids: tuple[str, ...] = Field(default_factory=tuple) | |
| determination_method: ResolutionDeterminationMethod | |
| rationale: str = Field(min_length=1) | |
| product_match_confidence: ConfidenceSignal | |
| review_required: bool | |
| downstream_classification_eligible: bool | |
| class CoverageModelResponseV3(DatasetModel): | |
| """Strict v3 coverage advice without pipeline-continuation authority.""" | |
| coverage_mode: CoverageMode | |
| qualifying_block_ids: tuple[str, ...] = Field(default_factory=tuple) | |
| rejected_block_ids: tuple[str, ...] = Field(default_factory=tuple) | |
| determination_method: ResolutionDeterminationMethod | |
| rationale: str = Field(min_length=1) | |
| product_match_confidence: ConfidenceSignal | |
| review_required: bool | |
| CoverageModelResponse = CoverageModelResponseV3 | |
| class ProductBlockModelDecision(DatasetModel): | |
| """Strict model decision for one supplied extraction block.""" | |
| block_id: str = Field(min_length=1) | |
| document_id: str = Field(min_length=1) | |
| page_number: int = Field(ge=1) | |
| section_id: str = Field(min_length=1) | |
| scope_type: ProductScopeType | |
| eligibility: EvidenceEligibility | |
| applicable_products: tuple[str, ...] = Field(default_factory=tuple) | |
| family_label: str | None = None | |
| qualifying_block_ids: tuple[str, ...] = Field(default_factory=tuple) | |
| declaration_ids: tuple[str, ...] = Field(default_factory=tuple) | |
| aliases_used: tuple[str, ...] = Field(default_factory=tuple) | |
| determination_method: ResolutionDeterminationMethod | |
| rationale: str = Field(min_length=1) | |
| product_match_confidence: ConfidenceSignal | |
| review_required: bool | |
| warnings: tuple[str, ...] = Field(default_factory=tuple) | |
| class ProductResolutionModelResponse(DatasetModel): | |
| """Strict versioned dataset product-resolution batch response.""" | |
| decisions: tuple[ProductBlockModelDecision, ...] | |
| class ProductPolicyViolationDiagnostic(DatasetModel): | |
| """Sanitized facts identifying one deterministic product-policy violation.""" | |
| rule_id: Literal[ | |
| "EXACT_TARGET_REQUIRES_ELIGIBLE", | |
| "EXACT_TARGET_REQUIRES_TARGET_SHORT_NAME", | |
| "ENUMERATED_TARGET_REQUIRES_ELIGIBLE", | |
| "ENUMERATED_TARGET_REQUIRES_TARGET_SHORT_NAME", | |
| "OTHER_PRODUCT_REQUIRES_EXCLUDED", | |
| "AMBIGUOUS_REQUIRES_REVIEW_ONLY", | |
| "MIXED_PRODUCT_REQUIRES_REVIEW_ONLY", | |
| "ELIGIBLE_FAMILY_REQUIRES_FAMILY_DECLARATION", | |
| "ELIGIBLE_FAMILY_REQUIRES_TARGET_MEMBERSHIP", | |
| "DOCUMENT_CONTEXT_ELIGIBILITY_INVALID", | |
| ] | |
| block_id: str = Field(min_length=1) | |
| scope_type: ProductScopeType | |
| eligibility: EvidenceEligibility | |
| eligibility_valid: bool | |
| target_short_name_present: bool | None = None | |
| family_declaration_present: bool | None = None | |
| target_membership_present: bool | None = None | |
| class ProductResolutionAttemptDiagnostic(DatasetModel): | |
| """Sanitized metadata for one rejected coverage or product-resolution attempt.""" | |
| phase: Literal["coverage", "product_resolution"] | |
| batch_index: int | None = Field(default=None, ge=1) | |
| total_batch_count: int | None = Field(default=None, ge=1) | |
| block_count: int = Field(ge=0) | |
| total_block_characters: int = Field(ge=0) | |
| signal_count: int = Field(ge=0) | |
| declaration_count: int = Field(ge=0) | |
| attempt_number: int = Field(ge=1) | |
| maximum_attempts: int = Field(ge=1) | |
| failure_code: str = Field(min_length=1) | |
| failure_category: Literal["validation", "schema", "provider"] | |
| policy_violation: ProductPolicyViolationDiagnostic | None = None | |
| def validate_attempt_context(self) -> ProductResolutionAttemptDiagnostic: | |
| if self.attempt_number > self.maximum_attempts: | |
| raise ValueError("attempt_number cannot exceed maximum_attempts") | |
| if self.phase == "coverage": | |
| if self.batch_index is not None or self.total_batch_count is not None: | |
| raise ValueError("coverage diagnostics cannot contain batch indices") | |
| elif ( | |
| self.batch_index is None | |
| or self.total_batch_count is None | |
| or self.batch_index > self.total_batch_count | |
| ): | |
| raise ValueError("product-resolution diagnostics require a valid batch index") | |
| if self.failure_code == "PRODUCT_POLICY_INVALID": | |
| if self.policy_violation is None: | |
| raise ValueError("product-policy failures require a sanitized subrule diagnostic") | |
| elif self.policy_violation is not None: | |
| raise ValueError("policy_violation is only valid for product-policy failures") | |
| return self | |
| class ProductBlockDecision(DatasetModel): | |
| """Validated authoritative linkage for one block-level scope decision.""" | |
| block_id: str = Field(min_length=1) | |
| document_id: str = Field(min_length=1) | |
| page_number: int = Field(ge=1) | |
| section_id: str = Field(min_length=1) | |
| source_order: int = Field(ge=0) | |
| scope_type: ProductScopeType | |
| eligibility: EvidenceEligibility | |
| applicable_products: tuple[str, ...] = Field(default_factory=tuple) | |
| family_label: str | None = None | |
| qualifying_block_ids: tuple[str, ...] = Field(default_factory=tuple) | |
| declaration_ids: tuple[str, ...] = Field(default_factory=tuple) | |
| determination_method: ResolutionDeterminationMethod | |
| rule_version: str = Field(min_length=1) | |
| prompt_version: str = Field(min_length=1) | |
| rationale: str = Field(min_length=1) | |
| product_match_confidence: ConfidenceSignal | |
| review_required: bool | |
| review_reasons: tuple[ProductReviewReason, ...] = Field(default_factory=tuple) | |
| warnings: tuple[DatasetWarning, ...] = Field(default_factory=tuple) | |
| class DatasetCoverageResolution(DatasetModel): | |
| """Complete blind coverage and product-scope output before evidence selection.""" | |
| target: DatasetIdentity | |
| extraction_manifest_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| selected_candidate_id: str = Field(min_length=1) | |
| selected_readme_url: str = Field(min_length=1) | |
| pdf_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| deterministic_analysis_version: str = Field(min_length=1) | |
| coverage_prompt_version: Literal[ | |
| "dataset_coverage_v1", "dataset_coverage_v2", "dataset_coverage_v3" | |
| ] | |
| product_resolution_prompt_version: Literal[ | |
| "dataset_product_resolution_v1", | |
| "dataset_product_resolution_v2", | |
| "dataset_product_resolution_v3", | |
| ] | |
| response_schema_version: Literal[ | |
| "dataset_product_resolution_schema_v1", | |
| "dataset_product_resolution_schema_v2", | |
| "dataset_product_resolution_schema_v3", | |
| ] | |
| declaration_locality_policy_version: str | None = None | |
| downstream_eligibility_policy_version: str | None = None | |
| review_aggregation_policy_version: str | None = None | |
| mixed_product_transition_policy_version: str | None = None | |
| signals: tuple[DeterministicProductSignal, ...] | |
| declarations: tuple[ProductDeclaration, ...] | |
| aliases: tuple[ProductAliasProvenance, ...] | |
| coverage_mode: CoverageMode | |
| qualifying_block_ids: tuple[str, ...] | |
| rejected_block_ids: tuple[str, ...] | |
| determination_method: ResolutionDeterminationMethod | |
| rationale: str = Field(min_length=1) | |
| product_match_confidence: ConfidenceSignal | |
| review_required: bool | |
| review_reasons: tuple[ProductReviewReason, ...] = Field(default_factory=tuple) | |
| uncertain_block_ids: tuple[str, ...] = Field(default_factory=tuple) | |
| downstream_classification_eligible: bool | |
| block_decisions: tuple[ProductBlockDecision, ...] | |
| target_eligible_block_ids: tuple[str, ...] | |
| excluded_other_product_block_ids: tuple[str, ...] | |
| ambiguous_review_block_ids: tuple[str, ...] | |
| document_context_block_ids: tuple[str, ...] | |
| warnings: tuple[DatasetWarning, ...] = Field(default_factory=tuple) | |
| blind_cache_identity_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| model_calls: int = Field(ge=0) | |
| def validate_block_partition(self) -> DatasetCoverageResolution: | |
| decisions = list(self.block_decisions) | |
| if [decision.source_order for decision in decisions] != list(range(len(decisions))): | |
| raise ValueError("block decisions must preserve consecutive source order") | |
| ids = [decision.block_id for decision in decisions] | |
| if len(ids) != len(set(ids)): | |
| raise ValueError("block decisions must be unique") | |
| partitions = ( | |
| self.target_eligible_block_ids, | |
| self.excluded_other_product_block_ids, | |
| self.ambiguous_review_block_ids, | |
| self.document_context_block_ids, | |
| ) | |
| flattened = [item for group in partitions for item in group] | |
| if len(flattened) != len(set(flattened)) or set(flattened) != set(ids): | |
| raise ValueError("eligibility collections must exactly partition block decisions") | |
| if not set(self.uncertain_block_ids).issubset(ids): | |
| raise ValueError("uncertain block IDs must reference block decisions") | |
| if self.response_schema_version == "dataset_product_resolution_schema_v3": | |
| policies = ( | |
| self.declaration_locality_policy_version, | |
| self.downstream_eligibility_policy_version, | |
| self.review_aggregation_policy_version, | |
| self.mixed_product_transition_policy_version, | |
| ) | |
| if any(value is None for value in policies): | |
| raise ValueError("v3 product resolution requires all policy versions") | |
| if any(item.locality is None for item in self.declarations): | |
| raise ValueError("v3 declarations require deterministic locality") | |
| return self | |
| class CoverageResolutionFailure(DatasetModel): | |
| """Typed pre-classification coverage failure after bounded retries.""" | |
| error: DatasetErrorRecord | |
| classification_attempted: Literal[False] = False | |
| gcmd_classifier_calls: Literal[0] = 0 | |
| downstream_classification_eligible: Literal[False] = False | |
| model_calls: int = Field(ge=0) | |
| attempt_diagnostics: tuple[ProductResolutionAttemptDiagnostic, ...] = Field( | |
| default_factory=tuple | |
| ) | |
| coverage_result: Literal[None] = None | |
| comparison: Literal[None] = None | |
| class EvidenceGroupMember(DatasetModel): | |
| """Traceable membership of one authoritative block in one group.""" | |
| block_id: str = Field(min_length=1) | |
| membership: GroupMembership | |
| linked_target_block_ids: tuple[str, ...] = Field(default_factory=tuple) | |
| class EvidenceBlockGroup(DatasetModel): | |
| """Deterministic bounded source-ordered evidence-selection group.""" | |
| group_id: str = Field(pattern=r"^group-[0-9]{4}$") | |
| extraction_manifest_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| grouping_policy_version: str = Field(min_length=1) | |
| members: tuple[EvidenceGroupMember, ...] | |
| source_character_count: int = Field(ge=0) | |
| max_blocks: int = Field(gt=0) | |
| max_characters: int = Field(gt=0) | |
| context_link_method: str = Field(min_length=1) | |
| class EvidenceSelectionInput(DatasetModel): | |
| """Minimum blind typed input accepted by the evidence stage.""" | |
| extraction_manifest: DatasetExtractionManifest | |
| product_resolution: DatasetCoverageResolution | |
| def validate_lineage(self) -> EvidenceSelectionInput: | |
| if self.extraction_manifest.lineage.identity != self.product_resolution.target: | |
| raise ValueError("evidence target must match extraction and product resolution") | |
| if self.extraction_manifest.document.sha256 != self.product_resolution.pdf_sha256: | |
| raise ValueError("evidence PDF identity must match product resolution") | |
| return self | |
| class EvidenceSelectionAudit(DatasetModel): | |
| """Derived selection metadata retained separately from packet source text.""" | |
| grouping_policy_version: str = Field(min_length=1) | |
| prompt_version: str = Field(min_length=1) | |
| response_schema_version: str = Field(min_length=1) | |
| groups: tuple[EvidenceBlockGroup, ...] | |
| decisions: tuple[EvidenceSelectionDecision | EvidenceSelectionDecisionV1, ...] | |
| excluded_decisions: tuple[ProductBlockDecision, ...] | |
| ambiguous_decisions: tuple[ProductBlockDecision, ...] | |
| model_calls: int = Field(ge=0) | |
| selection_model_calls: int = Field(default=0, ge=0) | |
| minimality_model_calls: int = Field(default=0, ge=0) | |
| minimality_policy_version: str = Field(default="dataset_evidence_minimization_v1") | |
| minimality_ran: bool = False | |
| candidate_block_count: int = Field(default=0, ge=0) | |
| initially_selected_block_count: int = Field(default=0, ge=0) | |
| final_independent_block_count: int = Field(default=0, ge=0) | |
| final_context_block_count: int = Field(default=0, ge=0) | |
| retained_block_ids: tuple[str, ...] = Field(default_factory=tuple) | |
| removed_block_ids: tuple[str, ...] = Field(default_factory=tuple) | |
| exclusion_records: tuple[EvidenceExclusionRecord, ...] = Field(default_factory=tuple) | |
| blind_cache_identity_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") | |
| class EvidencePacketResult(DatasetModel): | |
| """Successful Milestone 6 result without any GCMD classification.""" | |
| packet: DatasetEvidencePacket | |
| selection_audit: EvidenceSelectionAudit | |
| classification_attempted: Literal[False] = False | |
| gcmd_classifier_calls: Literal[0] = 0 | |
| comparison: Literal[None] = None | |
| class EvidenceNoUsableScienceOutcome(DatasetModel): | |
| """Successful semantic stop before the GCMD classifier boundary.""" | |
| reason_code: Literal["NO_USABLE_SCIENCE_EVIDENCE"] = "NO_USABLE_SCIENCE_EVIDENCE" | |
| processing_status: Literal[DatasetProcessingStatus.COMPLETED] = ( | |
| DatasetProcessingStatus.COMPLETED | |
| ) | |
| classification_outcome: Literal[DatasetClassificationOutcome.NOT_CLASSIFIED] = ( | |
| DatasetClassificationOutcome.NOT_CLASSIFIED | |
| ) | |
| no_classification_reason: str = "No independently usable target-applicable science evidence." | |
| selection_audit: EvidenceSelectionAudit | |
| classification_attempted: Literal[False] = False | |
| gcmd_classifier_calls: Literal[0] = 0 | |
| packet: Literal[None] = None | |
| comparison: Literal[None] = None | |
| class EvidenceSelectionFailure(DatasetModel): | |
| """Typed evidence-stage stop after validation or bounded retry failure.""" | |
| error: DatasetErrorRecord | |
| review_status: DatasetReviewStatus = DatasetReviewStatus.PENDING | |
| classification_attempted: Literal[False] = False | |
| gcmd_classifier_calls: Literal[0] = 0 | |
| model_calls: int = Field(ge=0) | |
| packet: Literal[None] = None | |
| comparison: Literal[None] = None | |
| class DatasetClassifierInput(DatasetModel): | |
| """Minimal future boundary: a validated packet is the only README text input.""" | |
| target: DatasetIdentity | |
| evidence_packet: DatasetEvidencePacket | |
| def validate_packet_target(self) -> DatasetClassifierInput: | |
| if self.target != self.evidence_packet.target: | |
| raise ValueError("classifier target must match the validated evidence packet") | |
| if self.evidence_packet.packet_readiness is not PacketReadiness.CLASSIFICATION_READY: | |
| raise ValueError("review-required packets cannot cross the classifier boundary") | |
| return self | |