ishaq101's picture
/fix parsing and term extract (#21)
f07443e
Raw History Blame Contribute Delete
20.4 kB
"""Pydantic contracts for the knowledge-extraction pipeline.
Three invariants are encoded here rather than described in prose, because every
one of them is a control that a later change could quietly remove:
1. **All content fields are Optional.** A model that cannot answer null will
fabricate one. Abstention is correct behaviour, never an error.
2. **`subdomain_tags` is an enum.** Classification, not generation.
3. **`Provenance.span` is mandatory and verbatim-checked.** It is the primary
anti-hallucination control and the thing that makes expert review
finishable — the reviewer checks a quote against a page, not a claim
against their memory.
`Chunk` here is the pipeline's **internal** unit, deliberately narrower than the
parsed-document artifact being agreed with Sofhia (the seam). Stages depend only
on this subset; `adapter.py` maps the seam type onto it, so seam churn lands in
one file instead of seven. See KNOWLEDGE_PIPELINE_TODO.md §3.
"""
from __future__ import annotations
from enum import Enum
from typing import Literal
from pydantic import BaseModel, Field, computed_field
Branch = Literal["glossary", "rule", "formula", "summary"]
ExtractionStatus = Literal["ok", "no_definition_found", "escalated"]
DiffStatus = Literal["new", "duplicate", "conflicting"]
RuleType = Literal["interpretation", "calculation"]
# Which guarantee a stored `formula_latex` actually carries. Three different
# paths leave the field populated and, until this existed, the output could not
# tell them apart — so a verified formula and an unverifiable one reached the
# expert as identical rows.
LatexVerification = Literal[
# Located in the source markup. The strong guarantee.
"verified",
# The artifact carried no `latex` for this chunk — written before the field
# crossed the seam, or a parser that emits none. Guarded by provenance only.
"unverified_no_markup",
# The source cannot express the operator being claimed: the `pipeline`
# backend writes multiplication as a bare letter `x`, so a claim spelling
# out `\times` is unprovable rather than wrong. Guarded by provenance only.
"unverified_operator",
]
class SubdomainEnum(str, Enum):
"""Classification target. Extend deliberately — a new member changes what
the model is allowed to answer, which is a prompt change, not a data one."""
production = "production"
maintenance = "maintenance"
hauling = "hauling"
loading = "loading"
drilling_blasting = "drilling_blasting"
equipment = "equipment"
safety = "safety"
quality = "quality"
planning = "planning"
cost = "cost"
geology = "geology"
other = "other"
# ── Stage 1: the chunk (internal view of the seam artifact) ─────────────
class ChunkAsset(BaseModel):
"""A figure or table referenced by a chunk. Crosses the seam at 0.4.0.
The subset of the parsing half's `Asset` that extraction actually uses. The
split between the two content fields is the whole point and must be preserved
into the prompt:
- `caption` is **verbatim document text**, so it is quotable as evidence and
the span check can locate it.
- `description` is **model-written** (a vision model's claim about the
picture). It is never span-checkable, never evidence, and must never be
copied into `Chunk.text` — `text` is the haystack the span check searches,
so generated prose in there would let a fabrication pass the control built
to catch it.
"""
asset_id: str
kind: str
caption: str | None = None
description: str | None = None
page_no: int | None = None
storage_key: str | None = None
class Chunk(BaseModel):
"""One unit of a parsed document, as the extraction stages need it.
`text` must stay **verbatim** from the source document. Span validation
locates LLM-quoted spans literally inside this text; if it is ever reflowed
or whitespace-normalised the lookup fails and the field is silently set to
null. The failure presents as a bad model, but the cause would be here.
"""
chunk_id: str
doc_id: str
text: str
page_start: int
page_end: int
ordinal: int = 0
# Structural context. Both Optional — many documents carry no numbering.
section_no: str | None = None
heading: str | None = None
# Breadcrumb of enclosing headings, outermost first, carried verbatim from
# the artifact. Empty is normal: a document of mid-chapter pages with no
# headings yields none, and fixtures written before this crossed the seam
# carry nothing.
#
# The `DocumentBrief.outline` is derived from these, which is the whole
# reason they now cross: the document's own structure is the one piece of
# domain knowledge that needs no model at all, and extraction was dropping
# it on the floor at the adapter — the same way it once dropped `latex`.
heading_path: list[str] = Field(default_factory=list)
# Figures and tables this chunk carries or points at. New at schema 0.4.0.
# Before it, a figure was its own chunk with EMPTY text, which made it
# unreachable: no text -> no mentions -> no cluster -> never evidence. 13
# figures across three documents, each described by a vision model we paid
# for, and none ever reached a prompt.
assets: list[ChunkAsset] = Field(default_factory=list)
# chunk_ids pointing AT this chunk (set on table chunks, which stay
# standalone because their linearised text earns its evidence slot).
referenced_by: list[str] = Field(default_factory=list)
# Cheap downstream filters / ranking signals
has_formula: bool = False
is_tabular: bool = False
bold_spans: list[str] = Field(default_factory=list)
# Source markup for the formula branch, carried verbatim from the artifact.
#
# `text` holds a readable RENDERING of the formula, never the markup, because
# the term filter is an NER model reading prose. So a `formula_latex` claim
# cannot be proved against `text` — the two forms never match. This field is
# the haystack that claim is checked against instead.
#
# Empty is normal: artifacts written before this field existed carry no
# `latex`, and the span check treats that as "cannot verify" rather than
# "fails verification" — see `validate/span_check.py`.
latex: list[str] = Field(default_factory=list)
# Source table markup, kept beside the pipe-linearised `text`. Crosses the
# seam at 0.4.0; before that the parsing half emitted it and the adapter
# dropped it, so a symptom x cause matrix reached the model as a wall of
# pipe-delimited text with the row/column correspondence gone. `colspan` and
# `rowspan` live here, which is what a table branch needs to recover
# (row header, column header, cell) triples.
table_html: str | None = None
class ParsedDoc(BaseModel):
"""A document's chunks plus the identity needed to version and cache them."""
doc_id: str
source_ref: str
content_hash: str
n_pages: int
chunks: list[Chunk]
parser_name: str = "unknown"
parser_version: str = ""
# Which MinerU backend produced the text, discrete rather than folded into
# `parser_version`, because the CLI refuses to run without it. `pipeline` and
# `vlm` emit DIFFERENT TEXT from the same PDF — the same equation arrives as
# `\times` from one and a spaced literal `x` from the other — so two results
# measured on different backends are not comparable, and nothing downstream
# can tell them apart after the fact. Empty means the artifact did not say.
parser_backend: str = ""
used_heading_split: bool = False
# ── Stage 2: filters ────────────────────────────────────────────────────
class Mention(BaseModel):
"""One occurrence of a candidate term inside a chunk."""
surface: str
chunk_id: str
char_start: int
char_end: int
label: str = ""
score: float = 0.0
hit_span_cap: bool = False
class RuleCandidate(BaseModel):
"""A passage a discourse cue marks as possibly stating a rule of thumb."""
chunk_id: str
cue: str
char_start: int
char_end: int
snippet: str
class AbbrevPair(BaseModel):
"""`PA` ↔ `Physical Availability`, harvested from a legend block.
Legend extraction must run before clustering: without these, an
abbreviation and its expansion cluster as two unrelated terms.
"""
abbrev: str
expansion: str
chunk_id: str
class FilterResult(BaseModel):
doc_id: str
mentions: list[Mention] = Field(default_factory=list)
rule_candidates: list[RuleCandidate] = Field(default_factory=list)
abbrev_pairs: list[AbbrevPair] = Field(default_factory=list)
# ── Stage 3: clusters ───────────────────────────────────────────────────
class TermCluster(BaseModel):
"""All mentions of one term. **The LLM call unit is the cluster**, not the
chunk and not the mention — that is what cuts expert review burden, and it
is also the only reason conflicting definitions can be detected at all
(they must arrive in the same call to be compared)."""
cluster_id: str
canonical: str
variants: list[str] = Field(default_factory=list)
mentions: list[Mention] = Field(default_factory=list)
mention_count: int = 0
merge_reasons: list[str] = Field(default_factory=list)
# Ranked best-first. The FULL list is kept, not just the top K —
# escalation consumes the tail.
evidence_chunk_ids: list[str] = Field(default_factory=list)
evidence_scores: list[float] = Field(default_factory=list)
class ClusterResult(BaseModel):
doc_id: str
clusters: list[TermCluster] = Field(default_factory=list)
n_mentions: int = 0
n_clusters: int = 0
compression_ratio: float = 0.0
# ── Stage 4+: extracted entries ─────────────────────────────────────────
class Provenance(BaseModel):
"""Where a claim came from. `span` is mandatory and must appear verbatim in
the evidence text; a field whose span cannot be located is rejected, never
repaired. A repaired span is an unfalsifiable claim."""
doc_id: str
span: str
# 0-BASED, exactly as the parser reports it. The 1-based number a human
# reads is `page_no` below.
page: int | None = None
section_no: str | None = None
chunk_id: str | None = None
@computed_field # type: ignore[prop-decorator]
@property
def page_no(self) -> int | None:
"""1-based page number, for the reviewer and the review UI (S6b).
Mirrors `knowledge_parsing.contracts.Chunk.page_no` deliberately, down to
being DERIVED rather than stored: a stored copy is a second truth that
can drift from `page`, and nothing downstream has to remember to set it.
The seam carried this on the parsing side from 2026-09-02 and extraction
dropped it at the adapter - the same way `heading_path` and `latex` were
each dropped there before.
`None` when `page` is unknown: a page number invented for an entry whose
page we do not have would send the reviewer to the wrong page, which is
worse than showing nothing.
"""
return None if self.page is None else self.page + 1
class GlossaryEntry(BaseModel):
# Deterministic and stable across re-runs — see `ids.py`. This is the key an
# expert's approve/edit/reject decision hangs on, so it must not move when a
# prompt is retuned.
term_id: str
term: str
full_name: str | None = None
# The literal wording as the document writes it, un-normalised. The BUMA
# standard heads its section "Physical of Availability (PA)" while the
# legend says "Physical Availability"; the discrepancy is surfaced to the
# expert rather than silently corrected.
source_wording: str | None = None
definition: str | None = None
# Reference, not a restatement. v1 carried `formula_latex` here as well as
# on FormulaEntry, which is two places for one truth to drift apart.
# Resolved deterministically by `link.py`; null when the formula branch did
# not extract one.
defining_formula_id: str | None = None
subdomain_tags: list[SubdomainEnum] = Field(default_factory=list)
mention_count: int = 0
provenance: Provenance
extraction_status: ExtractionStatus = "ok"
diff_status: DiffStatus | None = None
definition_conflict: bool = False
conflict_variants: list[str] = Field(default_factory=list)
class RuleEntry(BaseModel):
"""A rule of thumb stated by the document — the Interpretation Pack.
The artifact exists to improve analytics insight from expert interpretation
rules: interpretation logic, action benchmarks, and rules for when to
conclude or not conclude from data.
`rule_type` exists because the reference corpus does not contain that.
Measured against the gold set, **0 of 15 rules are interpretation rules** —
they are calculation conventions (`PA_COMPOSITE_WEIGHTED`), data-sourcing
rules (`PTY_PRODUCTION_SOURCE`) and unit conventions. That is a property of
the document type: a *Standard Parameter* defines how to **calculate** a
parameter, not how to **read** it. Splitting the two keeps the conventions
we do extract without pretending they are interpretation logic, and lets a
consumer wanting only interpretation filter on one field.
`condition` + `consequence` is the seed of the interpretation logic tree: a
rule stored as one paragraph cannot be attached to a skill later; a trigger
and its consequence can. The tree itself, action benchmarks and explicit
conclude/do-not-conclude verdicts are deliberately NOT modelled yet —
designing their schema against zero real examples would be guessing.
"""
rule_id: str
rule_type: RuleType = "calculation"
statement: str | None = None
condition: str | None = None
consequence: str | None = None
# Resolved by `link.py`, never by the model. Empty is a valid answer.
formula_ids: list[str] = Field(default_factory=list)
term_ids: list[str] = Field(default_factory=list)
provenance: Provenance
class FormulaVariable(BaseModel):
symbol: str
# Resolved by `link.py`. Best-effort and null is expected: a symbol is not
# always a legend abbreviation — the formula prompt's own worked example
# emits "Total Hours" and "Breakdown", neither of which need be a clustered
# term. `meaning` stays as the fallback for exactly those.
term_id: str | None = None
meaning: str | None = None
class FormulaEntry(BaseModel):
formula_id: str
name: str | None = None
formula_latex: str | None = None
variables: list[FormulaVariable] = Field(default_factory=list)
unit: str | None = None
provenance: Provenance
# Set by the span check, never by the model. `null` when there is no
# `formula_latex` to characterise — the model abstained, or the field was
# rejected outright and the rejection is recorded separately.
latex_verification: LatexVerification | None = None
class KeyParameter(BaseModel):
"""One parameter the document foregrounds, and the term it resolves to.
`surface` is what the model named; `term_id` is what corroborated it. The
pair is kept rather than collapsed because the *failure* is the interesting
case: a surface that resolves to nothing is a claim about the document that
no span-verified term supports, which is the closest thing this branch has
to a hallucination detector.
"""
surface: str
term_id: str | None = None
class DocumentBrief(BaseModel):
"""Whole-document domain context. The only branch that cannot be
span-checked — a plausible summary is indistinguishable from a correct one,
which is why it belongs on the larger model tier when one is available.
Renamed from `BriefContext` 2026-09-01. The name was the smaller half of the
problem: the workstream diagram has always called this box *Domain
Knowledge*, while the fields describe a short orientation card. The shape
below is still v2's — the v3 proposal that closes the gap (deriving the
document outline and subdomain coverage, and turning `purpose` from a
generated summary into a span-guarded verbatim quote) is
`docs/knowledge/KNOWLEDGE_DOMAIN_CONTEXT_V3.md`, and is NOT implemented.
`summary_md` and `scope` were dropped in v2 for that same reason: the
longest unverifiable prose on the least verifiable branch bought the least,
and `scope` overlapped `purpose`. What remains is what a reviewer needs to
orient before working the queue."""
brief_id: str
title: str | None = None
# The document's OWN statement of purpose, quoted verbatim - not a summary
# of it. v2 asked the model to compose this and could not check the result;
# asking it to locate one instead makes the same information span-guarded,
# which is what took this branch off the unverifiable list.
purpose_verbatim: str | None = None
# Ordered, de-duplicated heading breadcrumbs over the document's chunks —
# DERIVED, never asked of the model. It is the cheapest domain knowledge
# available and the only field here that carries the strong guarantee for
# free: it is verbatim source structure, so there is nothing to hallucinate.
outline: list[str] = Field(default_factory=list)
# v2 carried bare strings. Each surface now resolves to the glossary term it
# names, deterministically, in `link.py`. A `term_id` of None means the model
# named a parameter no extracted term corroborates — reported, never dropped
# and never guessed, and it earns its own review-queue row.
key_parameters: list[KeyParameter] = Field(default_factory=list)
# Which subdomains this document actually covers — AGGREGATED from the
# glossary entries' own `subdomain_tags`, never asked of the model. The
# classification already happened once, per term, with evidence in front of
# it; asking a second time at document level would be the same judgement
# made with less context and no way to check it.
#
# Ordered by how many terms carry each tag, descending, with an alphabetical
# tiebreak so the order is stable across runs.
subdomains: list[SubdomainEnum] = Field(default_factory=list)
# What the document yielded. Cheap orientation for a reviewer deciding
# whether a queue is worth opening, and the fastest way to spot a run that
# silently extracted nothing.
n_terms: int = 0
n_formulas: int = 0
n_rules: int = 0
provenance: Provenance
class CallUsage(BaseModel):
"""Per-call accounting. `cached_tokens` comes from the API and is never
modelled: caching does not engage below 1024 prompt tokens, so assuming it
would understate cost by ~10x on the input side."""
branch: Branch
deployment: str
tier: str = "nano"
# What actually served the call, as opposed to what we asked for.
# `deployment` is OUR name for it and never changes; these come back from
# the API and can change under a stable deployment name.
#
# Added after 2026-08-27, when identical input scored 6/7 one day and 0/12
# the next and there was no way to tell whether the model had moved
# underneath us. A result file without these is a measurement of an unknown.
model_version: str = ""
system_fingerprint: str = ""
prompt_tokens: int = 0
cached_tokens: int = 0
completion_tokens: int = 0
latency_s: float = 0.0
retries: int = 0
structured_output_mode: str = ""
simulated: bool = False
class RejectedField(BaseModel):
"""Audit row for a field the span check refused. Kept so a reviewer can see
what the control caught rather than only what it let through."""
entry_term: str
field: str
offending_value: str
reason: str
branch: Branch