"""Tunables for the knowledge-extraction pipeline. Every value here was calibrated on real documents and each one has a reason recorded in KNOWLEDGE_PIPELINE_CALIBRATION.md. Change them deliberately — most were arrived at by a measurement, and two of them (`FUZZY_MIN_LEN`, `CACHE_MIN_TOKENS`) fix bugs that are silent when reintroduced. Label and cue sets live in `config/*.yaml` so they can be tuned without a code change: label phrasing is the main recall lever and the filter is very sensitive to it. """ from __future__ import annotations from functools import lru_cache from pathlib import Path import yaml CONFIG_DIR = Path(__file__).resolve().parent / "config" # ── Term filter ───────────────────────────────────────────────────────── # Variant C beat both the English-default and Indonesian-phrasing label sets: # the other two missed the same class (mining activities and materials). LABELS_VARIANT = "broad" # 0.25, not 0.35: measured recall 0.854 @ 0.25 vs 0.658 @ 0.35 on `broad`. # Precision falls (0.41 vs 0.50) and that is the intended trade — the filter is # deliberately over-inclusive, clustering and ranking absorb the noise, and a # term the filter never proposes can never be recovered downstream. SPAN_SCORE_THRESHOLD = 0.25 # The span NER model truncates past ~384 of its own tokens and *warns rather # than failing*, so a long chunk silently loses its tail. Indonesian technical # prose subword-tokenises at roughly 2.5x, so 220-word windows still tripped the # cap; 130 did not. Chunks are fed as overlapping windows with offsets remapped. # # ⚠️ STAYS AT 130. A 2026-09-10 sweep appeared to show 256 was free — recall flat # across 130/180/256/320 (E1 0.878 at every size) while clusters fell 6%/23%/34% # across the three test documents. It was raised, then REVERTED the same day after # counting the truncation directly with GLiNER's own tokeniser: # # windows exceeding the 384-token cap, by WINDOW_WORDS # 130 180 200 256 320 # BUMA (ID) 1 1 1 1 1 <- pre-existing, see below # Komatsu (EN) 0 0 0 1 10 # Open Pit (EN) 0 4 11 46 49 <- 46 of 99 windows! # # At 256 nearly HALF of the Open Pit textbook's windows lose their tail. The # "flat recall" that justified the raise was measured on BUMA, the one document # whose truncation count does not change with window size — so the metric was # structurally blind to the cost, in exactly the way E1 was blind to the # clustering over-merge. The cluster reduction was not better grouping; it was # content disappearing. # # This vindicates the original reasoning rather than overturning it: the ~2.5x # Indonesian subword ratio was the motivating case, but dense ENGLISH technical # prose also trips the cap by 180. 130 is the only size measured safe on both # English documents. # # ⚠️ Separate, PRE-EXISTING finding from the same measurement: BUMA has one window # that exceeds the cap at EVERY size including 130 (409 tokens from 130 words — a # dense Indonesian chunk). One chunk in that document has always been losing its # tail. Not caused by this work and not fixed by it; a smaller window or a # per-chunk token budget would be the fix, and neither is measured yet. WINDOW_WORDS = 130 WINDOW_OVERLAP = 30 SPAN_TOKEN_CAP = 12 # ── The per-window SUBWORD budget (fixes the pre-existing over-cap window) ── # # `WINDOW_WORDS` counts WORDS; the model truncates on SUBWORDS, and the ratio # between them is a property of the text, not of the setting. Measured with # GLiNER's own tokeniser across all three documents, 2026-09-11: # # windows over the 384-token cap chars/subword (min / mean) # BUMA 31 1 1.57 / 3.43 # Komatsu 144 0 2.53 / 3.58 # Open Pit 193 0 2.79 / 4.10 # # One window in 368 is over, and it is the one the 2026-09-10 sweep found and # left unfixed: `STD_2026_006_MNO::0024`, 130 words and 738 characters, which # tokenise to 409 subwords. That chunk has ALWAYS been losing its tail, at every # window size including 130, and the model warns rather than failing so nothing # ever surfaced it. # # The fix measures instead of guessing: each window is checked against the # model's own tokeniser and split only if it does not fit. A character budget # would have to assume the worst ratio seen anywhere (1.57) and would therefore # split at ~600 characters — chopping up the other 367 windows, which fit # comfortably, to fix one. SPAN_SUBWORD_CAP = 384 # Split at 90% of the cap. The tokeniser count excludes the special tokens and # the label prompt GLiNER prepends, so measuring exactly to 384 still truncates. SPAN_SUBWORD_MARGIN = 0.9 # Used only when the tokeniser cannot be located on the model (a GLiNER version # change). 1.57 is the tightest ratio measured anywhere in the corpus, so this # over-splits rather than silently truncating — the safe direction. SPAN_FALLBACK_CHARS_PER_TOKEN = 1.57 # A split halves and recurses. 4 takes a 130-word window down to ~8 words, which # is past any plausible cap; the bound exists so a pathological input cannot # recurse indefinitely. SPAN_MAX_SPLIT_DEPTH = 4 # Flat (non-overlapping) span decoding. MEASURED AND KEPT 2026-09-10: nested # decoding (`flat_ner=False`) was tested on all three documents on the hypothesis # that flat mode was discarding the parent/child spans compound terms need. It # was not. Nested added 7-12% more mentions, delivered EXACTLY ZERO recall gain # (E1 0.878 either way, E1c identical too), and increased token-subset # over-merges by 40-80% (BUMA 15->24, Komatsu 18->32, Open Pit 97->134). # `span_filter` leaves this at the library default rather than passing it; this # constant documents the decision. Do not re-open without a new measurement. SPAN_FLAT_NER = True # ── Clustering ────────────────────────────────────────────────────────── # ⚠️ BOTH CONSTANTS BELOW ARE RETIRED as of 2026-09-10. The fuzzy matching pass # they configured is commented out at its call site in `cluster/cluster.py`; they # are still imported there because the `_fuzzy_match` function itself is parked # (comment-out-don't-delete), and it reads them. Changing them now affects # NOTHING — the pass does not run. # # Why it was retired, measured across three documents: `token_set_ratio` returns # exactly 100 when one token set is a subset of the other, so `FUZZY_THRESHOLD` # was inert and head nouns always absorbed their compounds. 131 of 140 merges # were subset matches (all wrong); of the 9 remaining, 6 were plain plurals — # now handled deterministically by `normalize(lemma=True)` — and 3 were distinct # metrics wrongly merged. Retiring the pass took subset-merges to 0 on all three # documents and lifted post-cluster gold recall 0.8293 -> 0.8537. # # `FUZZY_MIN_LEN` did its own job correctly to the end: it stopped "PA" and "UA" # colliding. It simply never gated multi-word English compounds, which is the # case that did the damage. FUZZY_THRESHOLD = 92 FUZZY_MIN_LEN = 5 # ── Evidence ranking ──────────────────────────────────────────────────── # 3 -> 5 on 2026-09-10 (Rifqi's call). How many top-ranked chunks each extraction # call sees. Three things this changes, recorded because none is obvious: # # 1. **Cost.** Evidence is the variable part of the prompt, so this is the single # biggest lever on spend per call after cluster count. Measured effect on the # three test documents: see CHANGELOG_v2.md. It also lowers the CACHE HIT RATE, # because the fixed cached prefix becomes a smaller share of each prompt. # # 2. **Escalation gets rarer but deeper.** `rounds_available` needs more than K # evidence chunks to fire at all, so at 5 a cluster needs 6+ (not 4+) to # escalate; when it does, a round covers ranks 6-10 instead of 4-6. On the three # test documents most clusters have 1-2 evidence chunks, so escalation fires # less often than before. # # 3. **It is NOT supported by the ranker measurement, and that is fine.** The # 2026-09-10 ceiling check found 16 of 16 gold definitions already at rank <=3 # (14 at rank 1), so on the BUMA standard ranks 4-5 add cost and no recall. The # case for 5 is the two ENGLISH documents, where definitions are dispersed # across pages and there is no gold set to measure with — a judgement that more # context beats a cheaper call on that document class. Revisit with a number if # a second gold set ever exists. EVIDENCE_K = 5 CUE_PROXIMITY_CHARS = 100 # How many LINKED partner chunks may be pulled into an evidence window on top of # the ranked K (0.4.0 two-way table pointers; added 2026-09-11). 1, not more. # # A table keeps its own chunk, and the prose that introduces it is a separate # chunk; `referenced_by` records the pair in both directions. Measured on the # three v2 artifacts, ranking only (free — no GLiNER, no spend): # # clusters whose top-5 touches a linked pair, by side # clusters touches a pair BOTH sides only ONE side # BUMA 83 31 0 31 # Komatsu 93 55 0 55 # Open Pit 427 70 0 70 # # **Zero clusters see both sides.** The ranker never co-selects a table and the # prose that names it, so the partner is genuinely unseen content rather than a # duplicate — and that prose is frequently the table's only title ("Troubleshooting # of engine (S-mode)" above a bare `S-1 | 6` grid). # # Cost AT THE CAP OF 1, measured on the same artifacts: evidence the model reads # +14.5% / +24.5% / +9.9%, which is +3.4k / +11.2k / +59.5k tokens, i.e. # **+1.6% / +4.1% / +4.2%** of each document's total v2 prompt spend (the fixed # cached prefix dominates). Open Pit moves from 71% to ~74% of the 2M ingest # ceiling. Clusters that actually gain a partner: 31/83, 55/93, 70/427. # # **Expansion happens in `top_k`, NOT in `rank_evidence`.** Two invariants depend # on that placement: # # 1. The full ranked list stays untouched, so nothing that reads # `evidence_chunk_ids` — `rounds_available` included — sees a different number. # Appending partners to the RANKED list would inflate the evidence count and # change escalation's arithmetic as a side effect. That is currently moot # (`MAX_ESCALATION_ROUNDS = 0` since 2026-09-11; escalation produced 0 # definitions in 48 calls) but the separation is what keeps this change # orthogonal to that one, and to whatever replaces it. # 2. `evidence_text` (the span check) and `evidence_block` (the prompt) both take # their ids from `top_k`, so expanding there keeps "what the model reads is # exactly what the span check searches" true for free. Expanding in the prompt # renderer instead would break it, and every quote from a partner chunk would # be nulled as unlocatable. Verified: 0 parity failures across all 603 clusters. # # Set to 0 to disable. Two honest scope notes: # # - The document this helps most (Komatsu, 59% of clusters gain a partner) is the # one that returns 0 definitions for GENRE reasons. This is groundwork for the # table branch more than a glossary win. # - The counts above were measured on the v2 artifacts and the RECORDED v2 filter # payloads, so they predate the subword window-splitting change. Mention and # cluster counts move with that change; the pairing structure does not. EVIDENCE_LINK_PARTNERS = 1 EVIDENCE_WEIGHTS: dict[str, float] = { "definitional_cue_near": 5.0, "term_in_heading": 4.0, "in_legend_block": 3.5, # A figure whose CAPTION names this term sits in this chunk. Built and # measured 2026-09-11, then set to **0.0 — OFF, because it measured as a # no-op.** The code stays because the A/B harness and the reasoning are worth # more than the three lines, and a diagram-heavy document could revive it. # # It was built to explain `definitions_citing_a_caption = 0` on all three # documents in the first paid v2 run. Two hypotheses, both now falsified: # # 1. "Captions don't reach the term filter." FALSE — a caption is part of # `Chunk.text`, so GLiNER already scans it and already produces mentions # from it (2 of 2 captions on BUMA, 3 of 8 on Open Pit). # 2. "Figure-bearing chunks don't get selected." FALSE — a captioned chunk is # already inside the top-5 for 22 of 83 BUMA clusters and 64 of 427 Open # Pit clusters. Turning this weight up to 2.5 changed the top-5 of **1 # cluster out of 603** across all three documents, and moved zero clusters # from "no figure" to "has a figure". # # The actual reason is simpler, and visible by reading them: **not one caption # in the corpus is a definition.** They are figure titles — "Figure 1.3. # Relative ability to influence costs (Lee, 1984)", "Gambar 2.1 Analisis # Gain/Loss Berdasarkan Setiap Parameter Produksi". A model declining to quote # one as a definition is CORRECT, not a retrieval failure. # # If revived, weight it BELOW `term_in_heading`: a heading naming a term means # "this section is about X"; a caption naming it means only "a figure about X # is here". "asset_caption_names_term": 0.0, "formula_present": 2.0, "bold_or_italic": 1.5, "first_occurrence": 1.0, "tabular_penalty": -3.0, } # ── Chunking ──────────────────────────────────────────────────────────── MAX_CHUNK_TOKENS = 1500 MAX_HEADING_LEN = 90 BOILERPLATE_MIN_FRAC = 0.6 # ── Extraction ────────────────────────────────────────────────────────── TEMPERATURE = 0.0 # OpenAI-family prompt caching does not engage AT ALL below this many prompt # tokens, so a shorter fixed prefix caches nothing and costs ~10x on input. The # measured hit rate at/above it was 54%. CACHE_MIN_TOKENS = 1024 # ── Validation ────────────────────────────────────────────────────────── # ⚠️ RETIRED 2026-09-11 by setting it to 0 — the loop and its call site are # untouched (comment-out-don't-delete), and restoring it is this one character. # # Finding 6 asked for a number either way. Here it is, counted off the six # persisted paid runs (v1 K=3 and v2 K=5, three documents): # # glossary calls clusters extra calls definitions FROM escalation # BUMA v1 69 66 3 0 # BUMA v2 83 83 0 0 # Komatsu v1 94 75 19 0 # Komatsu v2 102 93 9 0 # Open Pit v1 361 349 12 0 # Open Pit v2 430 425 5 0 # # 48 escalation calls across three documents and two configurations, and NOT ONE # of them produced a definition the first round had missed. `extraction_status` # is set to "escalated" only when a later round succeeds where round 0 failed, # and it appears zero times in every one of those six files. # # Why it fails is worth keeping, because it argues against reviving it naively: # escalation hands the model the NEXT k evidence chunks, ranked below the ones it # already declined. The 2026-09-10 ranker ceiling check found 16 of 18 gold # definitions at rank <=3 — so if the definition is anywhere, it was already in # round 0, and ranks 6-10 are where it is least likely to be. Raising EVIDENCE_K # to 5 made this worse in both directions at once: escalation needs 6+ evidence # chunks to fire at all, and the rounds it does reach are even further down. # # What would justify reviving it: a retry that changes the QUESTION (a different # prompt, a larger tier) rather than one that changes the evidence to lower-ranked # evidence. That is a different feature, not this constant. MAX_ESCALATION_ROUNDS = 0 CONFLICT_OVERLAP_THRESHOLD = 0.4 DUPLICATE_OVERLAP_THRESHOLD = 0.8 @lru_cache(maxsize=4) def load_yaml(name: str) -> dict: with open(CONFIG_DIR / name, encoding="utf-8") as fh: return yaml.safe_load(fh) def labels_for(variant: str = LABELS_VARIANT) -> tuple[list[str], float]: """Returns (labels, threshold) for a label variant.""" cfg = load_yaml("labels.yaml") labels = cfg.get(variant) or cfg.get(LABELS_VARIANT) or [] return list(labels), float(cfg.get("threshold", SPAN_SCORE_THRESHOLD))