File size: 20,359 Bytes
b68816f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
f07443e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
b68816f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
f07443e
 
 
 
 
 
 
 
 
 
b68816f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
f07443e
 
 
 
 
 
 
 
b68816f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
"""Pydantic contracts for the knowledge-extraction pipeline.

Three invariants are encoded here rather than described in prose, because every
one of them is a control that a later change could quietly remove:

1. **All content fields are Optional.** A model that cannot answer null will
   fabricate one. Abstention is correct behaviour, never an error.
2. **`subdomain_tags` is an enum.** Classification, not generation.
3. **`Provenance.span` is mandatory and verbatim-checked.** It is the primary
   anti-hallucination control and the thing that makes expert review
   finishable β€” the reviewer checks a quote against a page, not a claim
   against their memory.

`Chunk` here is the pipeline's **internal** unit, deliberately narrower than the
parsed-document artifact being agreed with Sofhia (the seam). Stages depend only
on this subset; `adapter.py` maps the seam type onto it, so seam churn lands in
one file instead of seven. See KNOWLEDGE_PIPELINE_TODO.md Β§3.
"""

from __future__ import annotations

from enum import Enum
from typing import Literal

from pydantic import BaseModel, Field, computed_field

Branch = Literal["glossary", "rule", "formula", "summary"]
ExtractionStatus = Literal["ok", "no_definition_found", "escalated"]
DiffStatus = Literal["new", "duplicate", "conflicting"]
RuleType = Literal["interpretation", "calculation"]

# Which guarantee a stored `formula_latex` actually carries. Three different
# paths leave the field populated and, until this existed, the output could not
# tell them apart β€” so a verified formula and an unverifiable one reached the
# expert as identical rows.
LatexVerification = Literal[
    # Located in the source markup. The strong guarantee.
    "verified",
    # The artifact carried no `latex` for this chunk β€” written before the field
    # crossed the seam, or a parser that emits none. Guarded by provenance only.
    "unverified_no_markup",
    # The source cannot express the operator being claimed: the `pipeline`
    # backend writes multiplication as a bare letter `x`, so a claim spelling
    # out `\times` is unprovable rather than wrong. Guarded by provenance only.
    "unverified_operator",
]


class SubdomainEnum(str, Enum):
    """Classification target. Extend deliberately β€” a new member changes what
    the model is allowed to answer, which is a prompt change, not a data one."""

    production = "production"
    maintenance = "maintenance"
    hauling = "hauling"
    loading = "loading"
    drilling_blasting = "drilling_blasting"
    equipment = "equipment"
    safety = "safety"
    quality = "quality"
    planning = "planning"
    cost = "cost"
    geology = "geology"
    other = "other"


# ── Stage 1: the chunk (internal view of the seam artifact) ─────────────


class ChunkAsset(BaseModel):
    """A figure or table referenced by a chunk. Crosses the seam at 0.4.0.

    The subset of the parsing half's `Asset` that extraction actually uses. The
    split between the two content fields is the whole point and must be preserved
    into the prompt:

    - `caption` is **verbatim document text**, so it is quotable as evidence and
      the span check can locate it.
    - `description` is **model-written** (a vision model's claim about the
      picture). It is never span-checkable, never evidence, and must never be
      copied into `Chunk.text` β€” `text` is the haystack the span check searches,
      so generated prose in there would let a fabrication pass the control built
      to catch it.
    """

    asset_id: str
    kind: str
    caption: str | None = None
    description: str | None = None
    page_no: int | None = None
    storage_key: str | None = None


class Chunk(BaseModel):
    """One unit of a parsed document, as the extraction stages need it.

    `text` must stay **verbatim** from the source document. Span validation
    locates LLM-quoted spans literally inside this text; if it is ever reflowed
    or whitespace-normalised the lookup fails and the field is silently set to
    null. The failure presents as a bad model, but the cause would be here.
    """

    chunk_id: str
    doc_id: str
    text: str
    page_start: int
    page_end: int
    ordinal: int = 0

    # Structural context. Both Optional β€” many documents carry no numbering.
    section_no: str | None = None
    heading: str | None = None

    # Breadcrumb of enclosing headings, outermost first, carried verbatim from
    # the artifact. Empty is normal: a document of mid-chapter pages with no
    # headings yields none, and fixtures written before this crossed the seam
    # carry nothing.
    #
    # The `DocumentBrief.outline` is derived from these, which is the whole
    # reason they now cross: the document's own structure is the one piece of
    # domain knowledge that needs no model at all, and extraction was dropping
    # it on the floor at the adapter β€” the same way it once dropped `latex`.
    heading_path: list[str] = Field(default_factory=list)

    # Figures and tables this chunk carries or points at. New at schema 0.4.0.
    # Before it, a figure was its own chunk with EMPTY text, which made it
    # unreachable: no text -> no mentions -> no cluster -> never evidence. 13
    # figures across three documents, each described by a vision model we paid
    # for, and none ever reached a prompt.
    assets: list[ChunkAsset] = Field(default_factory=list)
    # chunk_ids pointing AT this chunk (set on table chunks, which stay
    # standalone because their linearised text earns its evidence slot).
    referenced_by: list[str] = Field(default_factory=list)

    # Cheap downstream filters / ranking signals
    has_formula: bool = False
    is_tabular: bool = False
    bold_spans: list[str] = Field(default_factory=list)

    # Source markup for the formula branch, carried verbatim from the artifact.
    #
    # `text` holds a readable RENDERING of the formula, never the markup, because
    # the term filter is an NER model reading prose. So a `formula_latex` claim
    # cannot be proved against `text` β€” the two forms never match. This field is
    # the haystack that claim is checked against instead.
    #
    # Empty is normal: artifacts written before this field existed carry no
    # `latex`, and the span check treats that as "cannot verify" rather than
    # "fails verification" β€” see `validate/span_check.py`.
    latex: list[str] = Field(default_factory=list)

    # Source table markup, kept beside the pipe-linearised `text`. Crosses the
    # seam at 0.4.0; before that the parsing half emitted it and the adapter
    # dropped it, so a symptom x cause matrix reached the model as a wall of
    # pipe-delimited text with the row/column correspondence gone. `colspan` and
    # `rowspan` live here, which is what a table branch needs to recover
    # (row header, column header, cell) triples.
    table_html: str | None = None


class ParsedDoc(BaseModel):
    """A document's chunks plus the identity needed to version and cache them."""

    doc_id: str
    source_ref: str
    content_hash: str
    n_pages: int
    chunks: list[Chunk]
    parser_name: str = "unknown"
    parser_version: str = ""

    # Which MinerU backend produced the text, discrete rather than folded into
    # `parser_version`, because the CLI refuses to run without it. `pipeline` and
    # `vlm` emit DIFFERENT TEXT from the same PDF β€” the same equation arrives as
    # `\times` from one and a spaced literal `x` from the other β€” so two results
    # measured on different backends are not comparable, and nothing downstream
    # can tell them apart after the fact. Empty means the artifact did not say.
    parser_backend: str = ""

    used_heading_split: bool = False


# ── Stage 2: filters ────────────────────────────────────────────────────


class Mention(BaseModel):
    """One occurrence of a candidate term inside a chunk."""

    surface: str
    chunk_id: str
    char_start: int
    char_end: int
    label: str = ""
    score: float = 0.0
    hit_span_cap: bool = False


class RuleCandidate(BaseModel):
    """A passage a discourse cue marks as possibly stating a rule of thumb."""

    chunk_id: str
    cue: str
    char_start: int
    char_end: int
    snippet: str


class AbbrevPair(BaseModel):
    """`PA` ↔ `Physical Availability`, harvested from a legend block.

    Legend extraction must run before clustering: without these, an
    abbreviation and its expansion cluster as two unrelated terms.
    """

    abbrev: str
    expansion: str
    chunk_id: str


class FilterResult(BaseModel):
    doc_id: str
    mentions: list[Mention] = Field(default_factory=list)
    rule_candidates: list[RuleCandidate] = Field(default_factory=list)
    abbrev_pairs: list[AbbrevPair] = Field(default_factory=list)


# ── Stage 3: clusters ───────────────────────────────────────────────────


class TermCluster(BaseModel):
    """All mentions of one term. **The LLM call unit is the cluster**, not the
    chunk and not the mention β€” that is what cuts expert review burden, and it
    is also the only reason conflicting definitions can be detected at all
    (they must arrive in the same call to be compared)."""

    cluster_id: str
    canonical: str
    variants: list[str] = Field(default_factory=list)
    mentions: list[Mention] = Field(default_factory=list)
    mention_count: int = 0
    merge_reasons: list[str] = Field(default_factory=list)

    # Ranked best-first. The FULL list is kept, not just the top K β€”
    # escalation consumes the tail.
    evidence_chunk_ids: list[str] = Field(default_factory=list)
    evidence_scores: list[float] = Field(default_factory=list)


class ClusterResult(BaseModel):
    doc_id: str
    clusters: list[TermCluster] = Field(default_factory=list)
    n_mentions: int = 0
    n_clusters: int = 0
    compression_ratio: float = 0.0


# ── Stage 4+: extracted entries ─────────────────────────────────────────


class Provenance(BaseModel):
    """Where a claim came from. `span` is mandatory and must appear verbatim in
    the evidence text; a field whose span cannot be located is rejected, never
    repaired. A repaired span is an unfalsifiable claim."""

    doc_id: str
    span: str

    # 0-BASED, exactly as the parser reports it. The 1-based number a human
    # reads is `page_no` below.
    page: int | None = None

    section_no: str | None = None
    chunk_id: str | None = None

    @computed_field  # type: ignore[prop-decorator]
    @property
    def page_no(self) -> int | None:
        """1-based page number, for the reviewer and the review UI (S6b).

        Mirrors `knowledge_parsing.contracts.Chunk.page_no` deliberately, down to
        being DERIVED rather than stored: a stored copy is a second truth that
        can drift from `page`, and nothing downstream has to remember to set it.
        The seam carried this on the parsing side from 2026-09-02 and extraction
        dropped it at the adapter - the same way `heading_path` and `latex` were
        each dropped there before.

        `None` when `page` is unknown: a page number invented for an entry whose
        page we do not have would send the reviewer to the wrong page, which is
        worse than showing nothing.
        """
        return None if self.page is None else self.page + 1


class GlossaryEntry(BaseModel):
    # Deterministic and stable across re-runs β€” see `ids.py`. This is the key an
    # expert's approve/edit/reject decision hangs on, so it must not move when a
    # prompt is retuned.
    term_id: str
    term: str
    full_name: str | None = None

    # The literal wording as the document writes it, un-normalised. The BUMA
    # standard heads its section "Physical of Availability (PA)" while the
    # legend says "Physical Availability"; the discrepancy is surfaced to the
    # expert rather than silently corrected.
    source_wording: str | None = None

    definition: str | None = None

    # Reference, not a restatement. v1 carried `formula_latex` here as well as
    # on FormulaEntry, which is two places for one truth to drift apart.
    # Resolved deterministically by `link.py`; null when the formula branch did
    # not extract one.
    defining_formula_id: str | None = None

    subdomain_tags: list[SubdomainEnum] = Field(default_factory=list)

    mention_count: int = 0
    provenance: Provenance
    extraction_status: ExtractionStatus = "ok"
    diff_status: DiffStatus | None = None
    definition_conflict: bool = False
    conflict_variants: list[str] = Field(default_factory=list)


class RuleEntry(BaseModel):
    """A rule of thumb stated by the document β€” the Interpretation Pack.

    The artifact exists to improve analytics insight from expert interpretation
    rules: interpretation logic, action benchmarks, and rules for when to
    conclude or not conclude from data.

    `rule_type` exists because the reference corpus does not contain that.
    Measured against the gold set, **0 of 15 rules are interpretation rules** β€”
    they are calculation conventions (`PA_COMPOSITE_WEIGHTED`), data-sourcing
    rules (`PTY_PRODUCTION_SOURCE`) and unit conventions. That is a property of
    the document type: a *Standard Parameter* defines how to **calculate** a
    parameter, not how to **read** it. Splitting the two keeps the conventions
    we do extract without pretending they are interpretation logic, and lets a
    consumer wanting only interpretation filter on one field.

    `condition` + `consequence` is the seed of the interpretation logic tree: a
    rule stored as one paragraph cannot be attached to a skill later; a trigger
    and its consequence can. The tree itself, action benchmarks and explicit
    conclude/do-not-conclude verdicts are deliberately NOT modelled yet β€”
    designing their schema against zero real examples would be guessing.
    """

    rule_id: str
    rule_type: RuleType = "calculation"
    statement: str | None = None
    condition: str | None = None
    consequence: str | None = None

    # Resolved by `link.py`, never by the model. Empty is a valid answer.
    formula_ids: list[str] = Field(default_factory=list)
    term_ids: list[str] = Field(default_factory=list)

    provenance: Provenance


class FormulaVariable(BaseModel):
    symbol: str
    # Resolved by `link.py`. Best-effort and null is expected: a symbol is not
    # always a legend abbreviation β€” the formula prompt's own worked example
    # emits "Total Hours" and "Breakdown", neither of which need be a clustered
    # term. `meaning` stays as the fallback for exactly those.
    term_id: str | None = None
    meaning: str | None = None


class FormulaEntry(BaseModel):
    formula_id: str
    name: str | None = None
    formula_latex: str | None = None
    variables: list[FormulaVariable] = Field(default_factory=list)
    unit: str | None = None
    provenance: Provenance

    # Set by the span check, never by the model. `null` when there is no
    # `formula_latex` to characterise β€” the model abstained, or the field was
    # rejected outright and the rejection is recorded separately.
    latex_verification: LatexVerification | None = None


class KeyParameter(BaseModel):
    """One parameter the document foregrounds, and the term it resolves to.

    `surface` is what the model named; `term_id` is what corroborated it. The
    pair is kept rather than collapsed because the *failure* is the interesting
    case: a surface that resolves to nothing is a claim about the document that
    no span-verified term supports, which is the closest thing this branch has
    to a hallucination detector.
    """

    surface: str
    term_id: str | None = None


class DocumentBrief(BaseModel):
    """Whole-document domain context. The only branch that cannot be
    span-checked β€” a plausible summary is indistinguishable from a correct one,
    which is why it belongs on the larger model tier when one is available.

    Renamed from `BriefContext` 2026-09-01. The name was the smaller half of the
    problem: the workstream diagram has always called this box *Domain
    Knowledge*, while the fields describe a short orientation card. The shape
    below is still v2's β€” the v3 proposal that closes the gap (deriving the
    document outline and subdomain coverage, and turning `purpose` from a
    generated summary into a span-guarded verbatim quote) is
    `docs/knowledge/KNOWLEDGE_DOMAIN_CONTEXT_V3.md`, and is NOT implemented.

    `summary_md` and `scope` were dropped in v2 for that same reason: the
    longest unverifiable prose on the least verifiable branch bought the least,
    and `scope` overlapped `purpose`. What remains is what a reviewer needs to
    orient before working the queue."""

    brief_id: str
    title: str | None = None

    # The document's OWN statement of purpose, quoted verbatim - not a summary
    # of it. v2 asked the model to compose this and could not check the result;
    # asking it to locate one instead makes the same information span-guarded,
    # which is what took this branch off the unverifiable list.
    purpose_verbatim: str | None = None

    # Ordered, de-duplicated heading breadcrumbs over the document's chunks β€”
    # DERIVED, never asked of the model. It is the cheapest domain knowledge
    # available and the only field here that carries the strong guarantee for
    # free: it is verbatim source structure, so there is nothing to hallucinate.
    outline: list[str] = Field(default_factory=list)

    # v2 carried bare strings. Each surface now resolves to the glossary term it
    # names, deterministically, in `link.py`. A `term_id` of None means the model
    # named a parameter no extracted term corroborates β€” reported, never dropped
    # and never guessed, and it earns its own review-queue row.
    key_parameters: list[KeyParameter] = Field(default_factory=list)

    # Which subdomains this document actually covers β€” AGGREGATED from the
    # glossary entries' own `subdomain_tags`, never asked of the model. The
    # classification already happened once, per term, with evidence in front of
    # it; asking a second time at document level would be the same judgement
    # made with less context and no way to check it.
    #
    # Ordered by how many terms carry each tag, descending, with an alphabetical
    # tiebreak so the order is stable across runs.
    subdomains: list[SubdomainEnum] = Field(default_factory=list)

    # What the document yielded. Cheap orientation for a reviewer deciding
    # whether a queue is worth opening, and the fastest way to spot a run that
    # silently extracted nothing.
    n_terms: int = 0
    n_formulas: int = 0
    n_rules: int = 0

    provenance: Provenance


class CallUsage(BaseModel):
    """Per-call accounting. `cached_tokens` comes from the API and is never
    modelled: caching does not engage below 1024 prompt tokens, so assuming it
    would understate cost by ~10x on the input side."""

    branch: Branch
    deployment: str
    tier: str = "nano"

    # What actually served the call, as opposed to what we asked for.
    # `deployment` is OUR name for it and never changes; these come back from
    # the API and can change under a stable deployment name.
    #
    # Added after 2026-08-27, when identical input scored 6/7 one day and 0/12
    # the next and there was no way to tell whether the model had moved
    # underneath us. A result file without these is a measurement of an unknown.
    model_version: str = ""
    system_fingerprint: str = ""

    prompt_tokens: int = 0
    cached_tokens: int = 0
    completion_tokens: int = 0
    latency_s: float = 0.0
    retries: int = 0
    structured_output_mode: str = ""
    simulated: bool = False


class RejectedField(BaseModel):
    """Audit row for a field the span check refused. Kept so a reviewer can see
    what the control caught rather than only what it let through."""

    entry_term: str
    field: str
    offending_value: str
    reason: str
    branch: Branch