Download src/knowledge_extraction/models.py from DataEyond/Agentic-Service-Data-Eyond-Catalog: direct link, hf CLI and curl.
- Browser
- Download file 20.4 kB
-
https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_extraction/models.py
- Command line
-
hf download hf://spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/src/knowledge_extraction/models.py
-
curl -L -o models.py https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_extraction/models.py
20.4 kB
| """Pydantic contracts for the knowledge-extraction pipeline. | |
| Three invariants are encoded here rather than described in prose, because every | |
| one of them is a control that a later change could quietly remove: | |
| 1. **All content fields are Optional.** A model that cannot answer null will | |
| fabricate one. Abstention is correct behaviour, never an error. | |
| 2. **`subdomain_tags` is an enum.** Classification, not generation. | |
| 3. **`Provenance.span` is mandatory and verbatim-checked.** It is the primary | |
| anti-hallucination control and the thing that makes expert review | |
| finishable — the reviewer checks a quote against a page, not a claim | |
| against their memory. | |
| `Chunk` here is the pipeline's **internal** unit, deliberately narrower than the | |
| parsed-document artifact being agreed with Sofhia (the seam). Stages depend only | |
| on this subset; `adapter.py` maps the seam type onto it, so seam churn lands in | |
| one file instead of seven. See KNOWLEDGE_PIPELINE_TODO.md §3. | |
| """ | |
| from __future__ import annotations | |
| from enum import Enum | |
| from typing import Literal | |
| from pydantic import BaseModel, Field, computed_field | |
| Branch = Literal["glossary", "rule", "formula", "summary"] | |
| ExtractionStatus = Literal["ok", "no_definition_found", "escalated"] | |
| DiffStatus = Literal["new", "duplicate", "conflicting"] | |
| RuleType = Literal["interpretation", "calculation"] | |
| # Which guarantee a stored `formula_latex` actually carries. Three different | |
| # paths leave the field populated and, until this existed, the output could not | |
| # tell them apart — so a verified formula and an unverifiable one reached the | |
| # expert as identical rows. | |
| LatexVerification = Literal[ | |
| # Located in the source markup. The strong guarantee. | |
| "verified", | |
| # The artifact carried no `latex` for this chunk — written before the field | |
| # crossed the seam, or a parser that emits none. Guarded by provenance only. | |
| "unverified_no_markup", | |
| # The source cannot express the operator being claimed: the `pipeline` | |
| # backend writes multiplication as a bare letter `x`, so a claim spelling | |
| # out `\times` is unprovable rather than wrong. Guarded by provenance only. | |
| "unverified_operator", | |
| ] | |
| class SubdomainEnum(str, Enum): | |
| """Classification target. Extend deliberately — a new member changes what | |
| the model is allowed to answer, which is a prompt change, not a data one.""" | |
| production = "production" | |
| maintenance = "maintenance" | |
| hauling = "hauling" | |
| loading = "loading" | |
| drilling_blasting = "drilling_blasting" | |
| equipment = "equipment" | |
| safety = "safety" | |
| quality = "quality" | |
| planning = "planning" | |
| cost = "cost" | |
| geology = "geology" | |
| other = "other" | |
| # ── Stage 1: the chunk (internal view of the seam artifact) ───────────── | |
| class ChunkAsset(BaseModel): | |
| """A figure or table referenced by a chunk. Crosses the seam at 0.4.0. | |
| The subset of the parsing half's `Asset` that extraction actually uses. The | |
| split between the two content fields is the whole point and must be preserved | |
| into the prompt: | |
| - `caption` is **verbatim document text**, so it is quotable as evidence and | |
| the span check can locate it. | |
| - `description` is **model-written** (a vision model's claim about the | |
| picture). It is never span-checkable, never evidence, and must never be | |
| copied into `Chunk.text` — `text` is the haystack the span check searches, | |
| so generated prose in there would let a fabrication pass the control built | |
| to catch it. | |
| """ | |
| asset_id: str | |
| kind: str | |
| caption: str | None = None | |
| description: str | None = None | |
| page_no: int | None = None | |
| storage_key: str | None = None | |
| class Chunk(BaseModel): | |
| """One unit of a parsed document, as the extraction stages need it. | |
| `text` must stay **verbatim** from the source document. Span validation | |
| locates LLM-quoted spans literally inside this text; if it is ever reflowed | |
| or whitespace-normalised the lookup fails and the field is silently set to | |
| null. The failure presents as a bad model, but the cause would be here. | |
| """ | |
| chunk_id: str | |
| doc_id: str | |
| text: str | |
| page_start: int | |
| page_end: int | |
| ordinal: int = 0 | |
| # Structural context. Both Optional — many documents carry no numbering. | |
| section_no: str | None = None | |
| heading: str | None = None | |
| # Breadcrumb of enclosing headings, outermost first, carried verbatim from | |
| # the artifact. Empty is normal: a document of mid-chapter pages with no | |
| # headings yields none, and fixtures written before this crossed the seam | |
| # carry nothing. | |
| # | |
| # The `DocumentBrief.outline` is derived from these, which is the whole | |
| # reason they now cross: the document's own structure is the one piece of | |
| # domain knowledge that needs no model at all, and extraction was dropping | |
| # it on the floor at the adapter — the same way it once dropped `latex`. | |
| heading_path: list[str] = Field(default_factory=list) | |
| # Figures and tables this chunk carries or points at. New at schema 0.4.0. | |
| # Before it, a figure was its own chunk with EMPTY text, which made it | |
| # unreachable: no text -> no mentions -> no cluster -> never evidence. 13 | |
| # figures across three documents, each described by a vision model we paid | |
| # for, and none ever reached a prompt. | |
| assets: list[ChunkAsset] = Field(default_factory=list) | |
| # chunk_ids pointing AT this chunk (set on table chunks, which stay | |
| # standalone because their linearised text earns its evidence slot). | |
| referenced_by: list[str] = Field(default_factory=list) | |
| # Cheap downstream filters / ranking signals | |
| has_formula: bool = False | |
| is_tabular: bool = False | |
| bold_spans: list[str] = Field(default_factory=list) | |
| # Source markup for the formula branch, carried verbatim from the artifact. | |
| # | |
| # `text` holds a readable RENDERING of the formula, never the markup, because | |
| # the term filter is an NER model reading prose. So a `formula_latex` claim | |
| # cannot be proved against `text` — the two forms never match. This field is | |
| # the haystack that claim is checked against instead. | |
| # | |
| # Empty is normal: artifacts written before this field existed carry no | |
| # `latex`, and the span check treats that as "cannot verify" rather than | |
| # "fails verification" — see `validate/span_check.py`. | |
| latex: list[str] = Field(default_factory=list) | |
| # Source table markup, kept beside the pipe-linearised `text`. Crosses the | |
| # seam at 0.4.0; before that the parsing half emitted it and the adapter | |
| # dropped it, so a symptom x cause matrix reached the model as a wall of | |
| # pipe-delimited text with the row/column correspondence gone. `colspan` and | |
| # `rowspan` live here, which is what a table branch needs to recover | |
| # (row header, column header, cell) triples. | |
| table_html: str | None = None | |
| class ParsedDoc(BaseModel): | |
| """A document's chunks plus the identity needed to version and cache them.""" | |
| doc_id: str | |
| source_ref: str | |
| content_hash: str | |
| n_pages: int | |
| chunks: list[Chunk] | |
| parser_name: str = "unknown" | |
| parser_version: str = "" | |
| # Which MinerU backend produced the text, discrete rather than folded into | |
| # `parser_version`, because the CLI refuses to run without it. `pipeline` and | |
| # `vlm` emit DIFFERENT TEXT from the same PDF — the same equation arrives as | |
| # `\times` from one and a spaced literal `x` from the other — so two results | |
| # measured on different backends are not comparable, and nothing downstream | |
| # can tell them apart after the fact. Empty means the artifact did not say. | |
| parser_backend: str = "" | |
| used_heading_split: bool = False | |
| # ── Stage 2: filters ──────────────────────────────────────────────────── | |
| class Mention(BaseModel): | |
| """One occurrence of a candidate term inside a chunk.""" | |
| surface: str | |
| chunk_id: str | |
| char_start: int | |
| char_end: int | |
| label: str = "" | |
| score: float = 0.0 | |
| hit_span_cap: bool = False | |
| class RuleCandidate(BaseModel): | |
| """A passage a discourse cue marks as possibly stating a rule of thumb.""" | |
| chunk_id: str | |
| cue: str | |
| char_start: int | |
| char_end: int | |
| snippet: str | |
| class AbbrevPair(BaseModel): | |
| """`PA` ↔ `Physical Availability`, harvested from a legend block. | |
| Legend extraction must run before clustering: without these, an | |
| abbreviation and its expansion cluster as two unrelated terms. | |
| """ | |
| abbrev: str | |
| expansion: str | |
| chunk_id: str | |
| class FilterResult(BaseModel): | |
| doc_id: str | |
| mentions: list[Mention] = Field(default_factory=list) | |
| rule_candidates: list[RuleCandidate] = Field(default_factory=list) | |
| abbrev_pairs: list[AbbrevPair] = Field(default_factory=list) | |
| # ── Stage 3: clusters ─────────────────────────────────────────────────── | |
| class TermCluster(BaseModel): | |
| """All mentions of one term. **The LLM call unit is the cluster**, not the | |
| chunk and not the mention — that is what cuts expert review burden, and it | |
| is also the only reason conflicting definitions can be detected at all | |
| (they must arrive in the same call to be compared).""" | |
| cluster_id: str | |
| canonical: str | |
| variants: list[str] = Field(default_factory=list) | |
| mentions: list[Mention] = Field(default_factory=list) | |
| mention_count: int = 0 | |
| merge_reasons: list[str] = Field(default_factory=list) | |
| # Ranked best-first. The FULL list is kept, not just the top K — | |
| # escalation consumes the tail. | |
| evidence_chunk_ids: list[str] = Field(default_factory=list) | |
| evidence_scores: list[float] = Field(default_factory=list) | |
| class ClusterResult(BaseModel): | |
| doc_id: str | |
| clusters: list[TermCluster] = Field(default_factory=list) | |
| n_mentions: int = 0 | |
| n_clusters: int = 0 | |
| compression_ratio: float = 0.0 | |
| # ── Stage 4+: extracted entries ───────────────────────────────────────── | |
| class Provenance(BaseModel): | |
| """Where a claim came from. `span` is mandatory and must appear verbatim in | |
| the evidence text; a field whose span cannot be located is rejected, never | |
| repaired. A repaired span is an unfalsifiable claim.""" | |
| doc_id: str | |
| span: str | |
| # 0-BASED, exactly as the parser reports it. The 1-based number a human | |
| # reads is `page_no` below. | |
| page: int | None = None | |
| section_no: str | None = None | |
| chunk_id: str | None = None | |
| # type: ignore[prop-decorator] | |
| def page_no(self) -> int | None: | |
| """1-based page number, for the reviewer and the review UI (S6b). | |
| Mirrors `knowledge_parsing.contracts.Chunk.page_no` deliberately, down to | |
| being DERIVED rather than stored: a stored copy is a second truth that | |
| can drift from `page`, and nothing downstream has to remember to set it. | |
| The seam carried this on the parsing side from 2026-09-02 and extraction | |
| dropped it at the adapter - the same way `heading_path` and `latex` were | |
| each dropped there before. | |
| `None` when `page` is unknown: a page number invented for an entry whose | |
| page we do not have would send the reviewer to the wrong page, which is | |
| worse than showing nothing. | |
| """ | |
| return None if self.page is None else self.page + 1 | |
| class GlossaryEntry(BaseModel): | |
| # Deterministic and stable across re-runs — see `ids.py`. This is the key an | |
| # expert's approve/edit/reject decision hangs on, so it must not move when a | |
| # prompt is retuned. | |
| term_id: str | |
| term: str | |
| full_name: str | None = None | |
| # The literal wording as the document writes it, un-normalised. The BUMA | |
| # standard heads its section "Physical of Availability (PA)" while the | |
| # legend says "Physical Availability"; the discrepancy is surfaced to the | |
| # expert rather than silently corrected. | |
| source_wording: str | None = None | |
| definition: str | None = None | |
| # Reference, not a restatement. v1 carried `formula_latex` here as well as | |
| # on FormulaEntry, which is two places for one truth to drift apart. | |
| # Resolved deterministically by `link.py`; null when the formula branch did | |
| # not extract one. | |
| defining_formula_id: str | None = None | |
| subdomain_tags: list[SubdomainEnum] = Field(default_factory=list) | |
| mention_count: int = 0 | |
| provenance: Provenance | |
| extraction_status: ExtractionStatus = "ok" | |
| diff_status: DiffStatus | None = None | |
| definition_conflict: bool = False | |
| conflict_variants: list[str] = Field(default_factory=list) | |
| class RuleEntry(BaseModel): | |
| """A rule of thumb stated by the document — the Interpretation Pack. | |
| The artifact exists to improve analytics insight from expert interpretation | |
| rules: interpretation logic, action benchmarks, and rules for when to | |
| conclude or not conclude from data. | |
| `rule_type` exists because the reference corpus does not contain that. | |
| Measured against the gold set, **0 of 15 rules are interpretation rules** — | |
| they are calculation conventions (`PA_COMPOSITE_WEIGHTED`), data-sourcing | |
| rules (`PTY_PRODUCTION_SOURCE`) and unit conventions. That is a property of | |
| the document type: a *Standard Parameter* defines how to **calculate** a | |
| parameter, not how to **read** it. Splitting the two keeps the conventions | |
| we do extract without pretending they are interpretation logic, and lets a | |
| consumer wanting only interpretation filter on one field. | |
| `condition` + `consequence` is the seed of the interpretation logic tree: a | |
| rule stored as one paragraph cannot be attached to a skill later; a trigger | |
| and its consequence can. The tree itself, action benchmarks and explicit | |
| conclude/do-not-conclude verdicts are deliberately NOT modelled yet — | |
| designing their schema against zero real examples would be guessing. | |
| """ | |
| rule_id: str | |
| rule_type: RuleType = "calculation" | |
| statement: str | None = None | |
| condition: str | None = None | |
| consequence: str | None = None | |
| # Resolved by `link.py`, never by the model. Empty is a valid answer. | |
| formula_ids: list[str] = Field(default_factory=list) | |
| term_ids: list[str] = Field(default_factory=list) | |
| provenance: Provenance | |
| class FormulaVariable(BaseModel): | |
| symbol: str | |
| # Resolved by `link.py`. Best-effort and null is expected: a symbol is not | |
| # always a legend abbreviation — the formula prompt's own worked example | |
| # emits "Total Hours" and "Breakdown", neither of which need be a clustered | |
| # term. `meaning` stays as the fallback for exactly those. | |
| term_id: str | None = None | |
| meaning: str | None = None | |
| class FormulaEntry(BaseModel): | |
| formula_id: str | |
| name: str | None = None | |
| formula_latex: str | None = None | |
| variables: list[FormulaVariable] = Field(default_factory=list) | |
| unit: str | None = None | |
| provenance: Provenance | |
| # Set by the span check, never by the model. `null` when there is no | |
| # `formula_latex` to characterise — the model abstained, or the field was | |
| # rejected outright and the rejection is recorded separately. | |
| latex_verification: LatexVerification | None = None | |
| class KeyParameter(BaseModel): | |
| """One parameter the document foregrounds, and the term it resolves to. | |
| `surface` is what the model named; `term_id` is what corroborated it. The | |
| pair is kept rather than collapsed because the *failure* is the interesting | |
| case: a surface that resolves to nothing is a claim about the document that | |
| no span-verified term supports, which is the closest thing this branch has | |
| to a hallucination detector. | |
| """ | |
| surface: str | |
| term_id: str | None = None | |
| class DocumentBrief(BaseModel): | |
| """Whole-document domain context. The only branch that cannot be | |
| span-checked — a plausible summary is indistinguishable from a correct one, | |
| which is why it belongs on the larger model tier when one is available. | |
| Renamed from `BriefContext` 2026-09-01. The name was the smaller half of the | |
| problem: the workstream diagram has always called this box *Domain | |
| Knowledge*, while the fields describe a short orientation card. The shape | |
| below is still v2's — the v3 proposal that closes the gap (deriving the | |
| document outline and subdomain coverage, and turning `purpose` from a | |
| generated summary into a span-guarded verbatim quote) is | |
| `docs/knowledge/KNOWLEDGE_DOMAIN_CONTEXT_V3.md`, and is NOT implemented. | |
| `summary_md` and `scope` were dropped in v2 for that same reason: the | |
| longest unverifiable prose on the least verifiable branch bought the least, | |
| and `scope` overlapped `purpose`. What remains is what a reviewer needs to | |
| orient before working the queue.""" | |
| brief_id: str | |
| title: str | None = None | |
| # The document's OWN statement of purpose, quoted verbatim - not a summary | |
| # of it. v2 asked the model to compose this and could not check the result; | |
| # asking it to locate one instead makes the same information span-guarded, | |
| # which is what took this branch off the unverifiable list. | |
| purpose_verbatim: str | None = None | |
| # Ordered, de-duplicated heading breadcrumbs over the document's chunks — | |
| # DERIVED, never asked of the model. It is the cheapest domain knowledge | |
| # available and the only field here that carries the strong guarantee for | |
| # free: it is verbatim source structure, so there is nothing to hallucinate. | |
| outline: list[str] = Field(default_factory=list) | |
| # v2 carried bare strings. Each surface now resolves to the glossary term it | |
| # names, deterministically, in `link.py`. A `term_id` of None means the model | |
| # named a parameter no extracted term corroborates — reported, never dropped | |
| # and never guessed, and it earns its own review-queue row. | |
| key_parameters: list[KeyParameter] = Field(default_factory=list) | |
| # Which subdomains this document actually covers — AGGREGATED from the | |
| # glossary entries' own `subdomain_tags`, never asked of the model. The | |
| # classification already happened once, per term, with evidence in front of | |
| # it; asking a second time at document level would be the same judgement | |
| # made with less context and no way to check it. | |
| # | |
| # Ordered by how many terms carry each tag, descending, with an alphabetical | |
| # tiebreak so the order is stable across runs. | |
| subdomains: list[SubdomainEnum] = Field(default_factory=list) | |
| # What the document yielded. Cheap orientation for a reviewer deciding | |
| # whether a queue is worth opening, and the fastest way to spot a run that | |
| # silently extracted nothing. | |
| n_terms: int = 0 | |
| n_formulas: int = 0 | |
| n_rules: int = 0 | |
| provenance: Provenance | |
| class CallUsage(BaseModel): | |
| """Per-call accounting. `cached_tokens` comes from the API and is never | |
| modelled: caching does not engage below 1024 prompt tokens, so assuming it | |
| would understate cost by ~10x on the input side.""" | |
| branch: Branch | |
| deployment: str | |
| tier: str = "nano" | |
| # What actually served the call, as opposed to what we asked for. | |
| # `deployment` is OUR name for it and never changes; these come back from | |
| # the API and can change under a stable deployment name. | |
| # | |
| # Added after 2026-08-27, when identical input scored 6/7 one day and 0/12 | |
| # the next and there was no way to tell whether the model had moved | |
| # underneath us. A result file without these is a measurement of an unknown. | |
| model_version: str = "" | |
| system_fingerprint: str = "" | |
| prompt_tokens: int = 0 | |
| cached_tokens: int = 0 | |
| completion_tokens: int = 0 | |
| latency_s: float = 0.0 | |
| retries: int = 0 | |
| structured_output_mode: str = "" | |
| simulated: bool = False | |
| class RejectedField(BaseModel): | |
| """Audit row for a field the span check refused. Kept so a reviewer can see | |
| what the control caught rather than only what it let through.""" | |
| entry_term: str | |
| field: str | |
| offending_value: str | |
| reason: str | |
| branch: Branch | |