File size: 17,365 Bytes
b68816f f07443e b68816f f07443e b68816f f07443e b68816f f07443e b68816f f07443e b68816f f07443e b68816f f07443e b68816f | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 | """Tunables for the knowledge-extraction pipeline.
Every value here was calibrated on real documents and each one has a reason
recorded in KNOWLEDGE_PIPELINE_CALIBRATION.md. Change them deliberately β most
were arrived at by a measurement, and two of them (`FUZZY_MIN_LEN`,
`CACHE_MIN_TOKENS`) fix bugs that are silent when reintroduced.
Label and cue sets live in `config/*.yaml` so they can be tuned without a code
change: label phrasing is the main recall lever and the filter is very sensitive
to it.
"""
from __future__ import annotations
from functools import lru_cache
from pathlib import Path
import yaml
CONFIG_DIR = Path(__file__).resolve().parent / "config"
# ββ Term filter βββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
# Variant C beat both the English-default and Indonesian-phrasing label sets:
# the other two missed the same class (mining activities and materials).
LABELS_VARIANT = "broad"
# 0.25, not 0.35: measured recall 0.854 @ 0.25 vs 0.658 @ 0.35 on `broad`.
# Precision falls (0.41 vs 0.50) and that is the intended trade β the filter is
# deliberately over-inclusive, clustering and ranking absorb the noise, and a
# term the filter never proposes can never be recovered downstream.
SPAN_SCORE_THRESHOLD = 0.25
# The span NER model truncates past ~384 of its own tokens and *warns rather
# than failing*, so a long chunk silently loses its tail. Indonesian technical
# prose subword-tokenises at roughly 2.5x, so 220-word windows still tripped the
# cap; 130 did not. Chunks are fed as overlapping windows with offsets remapped.
#
# β οΈ STAYS AT 130. A 2026-09-10 sweep appeared to show 256 was free β recall flat
# across 130/180/256/320 (E1 0.878 at every size) while clusters fell 6%/23%/34%
# across the three test documents. It was raised, then REVERTED the same day after
# counting the truncation directly with GLiNER's own tokeniser:
#
# windows exceeding the 384-token cap, by WINDOW_WORDS
# 130 180 200 256 320
# BUMA (ID) 1 1 1 1 1 <- pre-existing, see below
# Komatsu (EN) 0 0 0 1 10
# Open Pit (EN) 0 4 11 46 49 <- 46 of 99 windows!
#
# At 256 nearly HALF of the Open Pit textbook's windows lose their tail. The
# "flat recall" that justified the raise was measured on BUMA, the one document
# whose truncation count does not change with window size β so the metric was
# structurally blind to the cost, in exactly the way E1 was blind to the
# clustering over-merge. The cluster reduction was not better grouping; it was
# content disappearing.
#
# This vindicates the original reasoning rather than overturning it: the ~2.5x
# Indonesian subword ratio was the motivating case, but dense ENGLISH technical
# prose also trips the cap by 180. 130 is the only size measured safe on both
# English documents.
#
# β οΈ Separate, PRE-EXISTING finding from the same measurement: BUMA has one window
# that exceeds the cap at EVERY size including 130 (409 tokens from 130 words β a
# dense Indonesian chunk). One chunk in that document has always been losing its
# tail. Not caused by this work and not fixed by it; a smaller window or a
# per-chunk token budget would be the fix, and neither is measured yet.
WINDOW_WORDS = 130
WINDOW_OVERLAP = 30
SPAN_TOKEN_CAP = 12
# ββ The per-window SUBWORD budget (fixes the pre-existing over-cap window) ββ
#
# `WINDOW_WORDS` counts WORDS; the model truncates on SUBWORDS, and the ratio
# between them is a property of the text, not of the setting. Measured with
# GLiNER's own tokeniser across all three documents, 2026-09-11:
#
# windows over the 384-token cap chars/subword (min / mean)
# BUMA 31 1 1.57 / 3.43
# Komatsu 144 0 2.53 / 3.58
# Open Pit 193 0 2.79 / 4.10
#
# One window in 368 is over, and it is the one the 2026-09-10 sweep found and
# left unfixed: `STD_2026_006_MNO::0024`, 130 words and 738 characters, which
# tokenise to 409 subwords. That chunk has ALWAYS been losing its tail, at every
# window size including 130, and the model warns rather than failing so nothing
# ever surfaced it.
#
# The fix measures instead of guessing: each window is checked against the
# model's own tokeniser and split only if it does not fit. A character budget
# would have to assume the worst ratio seen anywhere (1.57) and would therefore
# split at ~600 characters β chopping up the other 367 windows, which fit
# comfortably, to fix one.
SPAN_SUBWORD_CAP = 384
# Split at 90% of the cap. The tokeniser count excludes the special tokens and
# the label prompt GLiNER prepends, so measuring exactly to 384 still truncates.
SPAN_SUBWORD_MARGIN = 0.9
# Used only when the tokeniser cannot be located on the model (a GLiNER version
# change). 1.57 is the tightest ratio measured anywhere in the corpus, so this
# over-splits rather than silently truncating β the safe direction.
SPAN_FALLBACK_CHARS_PER_TOKEN = 1.57
# A split halves and recurses. 4 takes a 130-word window down to ~8 words, which
# is past any plausible cap; the bound exists so a pathological input cannot
# recurse indefinitely.
SPAN_MAX_SPLIT_DEPTH = 4
# Flat (non-overlapping) span decoding. MEASURED AND KEPT 2026-09-10: nested
# decoding (`flat_ner=False`) was tested on all three documents on the hypothesis
# that flat mode was discarding the parent/child spans compound terms need. It
# was not. Nested added 7-12% more mentions, delivered EXACTLY ZERO recall gain
# (E1 0.878 either way, E1c identical too), and increased token-subset
# over-merges by 40-80% (BUMA 15->24, Komatsu 18->32, Open Pit 97->134).
# `span_filter` leaves this at the library default rather than passing it; this
# constant documents the decision. Do not re-open without a new measurement.
SPAN_FLAT_NER = True
# ββ Clustering ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
# β οΈ BOTH CONSTANTS BELOW ARE RETIRED as of 2026-09-10. The fuzzy matching pass
# they configured is commented out at its call site in `cluster/cluster.py`; they
# are still imported there because the `_fuzzy_match` function itself is parked
# (comment-out-don't-delete), and it reads them. Changing them now affects
# NOTHING β the pass does not run.
#
# Why it was retired, measured across three documents: `token_set_ratio` returns
# exactly 100 when one token set is a subset of the other, so `FUZZY_THRESHOLD`
# was inert and head nouns always absorbed their compounds. 131 of 140 merges
# were subset matches (all wrong); of the 9 remaining, 6 were plain plurals β
# now handled deterministically by `normalize(lemma=True)` β and 3 were distinct
# metrics wrongly merged. Retiring the pass took subset-merges to 0 on all three
# documents and lifted post-cluster gold recall 0.8293 -> 0.8537.
#
# `FUZZY_MIN_LEN` did its own job correctly to the end: it stopped "PA" and "UA"
# colliding. It simply never gated multi-word English compounds, which is the
# case that did the damage.
FUZZY_THRESHOLD = 92
FUZZY_MIN_LEN = 5
# ββ Evidence ranking ββββββββββββββββββββββββββββββββββββββββββββββββββββ
# 3 -> 5 on 2026-09-10 (Rifqi's call). How many top-ranked chunks each extraction
# call sees. Three things this changes, recorded because none is obvious:
#
# 1. **Cost.** Evidence is the variable part of the prompt, so this is the single
# biggest lever on spend per call after cluster count. Measured effect on the
# three test documents: see CHANGELOG_v2.md. It also lowers the CACHE HIT RATE,
# because the fixed cached prefix becomes a smaller share of each prompt.
#
# 2. **Escalation gets rarer but deeper.** `rounds_available` needs more than K
# evidence chunks to fire at all, so at 5 a cluster needs 6+ (not 4+) to
# escalate; when it does, a round covers ranks 6-10 instead of 4-6. On the three
# test documents most clusters have 1-2 evidence chunks, so escalation fires
# less often than before.
#
# 3. **It is NOT supported by the ranker measurement, and that is fine.** The
# 2026-09-10 ceiling check found 16 of 16 gold definitions already at rank <=3
# (14 at rank 1), so on the BUMA standard ranks 4-5 add cost and no recall. The
# case for 5 is the two ENGLISH documents, where definitions are dispersed
# across pages and there is no gold set to measure with β a judgement that more
# context beats a cheaper call on that document class. Revisit with a number if
# a second gold set ever exists.
EVIDENCE_K = 5
CUE_PROXIMITY_CHARS = 100
# How many LINKED partner chunks may be pulled into an evidence window on top of
# the ranked K (0.4.0 two-way table pointers; added 2026-09-11). 1, not more.
#
# A table keeps its own chunk, and the prose that introduces it is a separate
# chunk; `referenced_by` records the pair in both directions. Measured on the
# three v2 artifacts, ranking only (free β no GLiNER, no spend):
#
# clusters whose top-5 touches a linked pair, by side
# clusters touches a pair BOTH sides only ONE side
# BUMA 83 31 0 31
# Komatsu 93 55 0 55
# Open Pit 427 70 0 70
#
# **Zero clusters see both sides.** The ranker never co-selects a table and the
# prose that names it, so the partner is genuinely unseen content rather than a
# duplicate β and that prose is frequently the table's only title ("Troubleshooting
# of engine (S-mode)" above a bare `S-1 | 6` grid).
#
# Cost AT THE CAP OF 1, measured on the same artifacts: evidence the model reads
# +14.5% / +24.5% / +9.9%, which is +3.4k / +11.2k / +59.5k tokens, i.e.
# **+1.6% / +4.1% / +4.2%** of each document's total v2 prompt spend (the fixed
# cached prefix dominates). Open Pit moves from 71% to ~74% of the 2M ingest
# ceiling. Clusters that actually gain a partner: 31/83, 55/93, 70/427.
#
# **Expansion happens in `top_k`, NOT in `rank_evidence`.** Two invariants depend
# on that placement:
#
# 1. The full ranked list stays untouched, so nothing that reads
# `evidence_chunk_ids` β `rounds_available` included β sees a different number.
# Appending partners to the RANKED list would inflate the evidence count and
# change escalation's arithmetic as a side effect. That is currently moot
# (`MAX_ESCALATION_ROUNDS = 0` since 2026-09-11; escalation produced 0
# definitions in 48 calls) but the separation is what keeps this change
# orthogonal to that one, and to whatever replaces it.
# 2. `evidence_text` (the span check) and `evidence_block` (the prompt) both take
# their ids from `top_k`, so expanding there keeps "what the model reads is
# exactly what the span check searches" true for free. Expanding in the prompt
# renderer instead would break it, and every quote from a partner chunk would
# be nulled as unlocatable. Verified: 0 parity failures across all 603 clusters.
#
# Set to 0 to disable. Two honest scope notes:
#
# - The document this helps most (Komatsu, 59% of clusters gain a partner) is the
# one that returns 0 definitions for GENRE reasons. This is groundwork for the
# table branch more than a glossary win.
# - The counts above were measured on the v2 artifacts and the RECORDED v2 filter
# payloads, so they predate the subword window-splitting change. Mention and
# cluster counts move with that change; the pairing structure does not.
EVIDENCE_LINK_PARTNERS = 1
EVIDENCE_WEIGHTS: dict[str, float] = {
"definitional_cue_near": 5.0,
"term_in_heading": 4.0,
"in_legend_block": 3.5,
# A figure whose CAPTION names this term sits in this chunk. Built and
# measured 2026-09-11, then set to **0.0 β OFF, because it measured as a
# no-op.** The code stays because the A/B harness and the reasoning are worth
# more than the three lines, and a diagram-heavy document could revive it.
#
# It was built to explain `definitions_citing_a_caption = 0` on all three
# documents in the first paid v2 run. Two hypotheses, both now falsified:
#
# 1. "Captions don't reach the term filter." FALSE β a caption is part of
# `Chunk.text`, so GLiNER already scans it and already produces mentions
# from it (2 of 2 captions on BUMA, 3 of 8 on Open Pit).
# 2. "Figure-bearing chunks don't get selected." FALSE β a captioned chunk is
# already inside the top-5 for 22 of 83 BUMA clusters and 64 of 427 Open
# Pit clusters. Turning this weight up to 2.5 changed the top-5 of **1
# cluster out of 603** across all three documents, and moved zero clusters
# from "no figure" to "has a figure".
#
# The actual reason is simpler, and visible by reading them: **not one caption
# in the corpus is a definition.** They are figure titles β "Figure 1.3.
# Relative ability to influence costs (Lee, 1984)", "Gambar 2.1 Analisis
# Gain/Loss Berdasarkan Setiap Parameter Produksi". A model declining to quote
# one as a definition is CORRECT, not a retrieval failure.
#
# If revived, weight it BELOW `term_in_heading`: a heading naming a term means
# "this section is about X"; a caption naming it means only "a figure about X
# is here".
"asset_caption_names_term": 0.0,
"formula_present": 2.0,
"bold_or_italic": 1.5,
"first_occurrence": 1.0,
"tabular_penalty": -3.0,
}
# ββ Chunking ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
MAX_CHUNK_TOKENS = 1500
MAX_HEADING_LEN = 90
BOILERPLATE_MIN_FRAC = 0.6
# ββ Extraction ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
TEMPERATURE = 0.0
# OpenAI-family prompt caching does not engage AT ALL below this many prompt
# tokens, so a shorter fixed prefix caches nothing and costs ~10x on input. The
# measured hit rate at/above it was 54%.
CACHE_MIN_TOKENS = 1024
# ββ Validation ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
# β οΈ RETIRED 2026-09-11 by setting it to 0 β the loop and its call site are
# untouched (comment-out-don't-delete), and restoring it is this one character.
#
# Finding 6 asked for a number either way. Here it is, counted off the six
# persisted paid runs (v1 K=3 and v2 K=5, three documents):
#
# glossary calls clusters extra calls definitions FROM escalation
# BUMA v1 69 66 3 0
# BUMA v2 83 83 0 0
# Komatsu v1 94 75 19 0
# Komatsu v2 102 93 9 0
# Open Pit v1 361 349 12 0
# Open Pit v2 430 425 5 0
#
# 48 escalation calls across three documents and two configurations, and NOT ONE
# of them produced a definition the first round had missed. `extraction_status`
# is set to "escalated" only when a later round succeeds where round 0 failed,
# and it appears zero times in every one of those six files.
#
# Why it fails is worth keeping, because it argues against reviving it naively:
# escalation hands the model the NEXT k evidence chunks, ranked below the ones it
# already declined. The 2026-09-10 ranker ceiling check found 16 of 18 gold
# definitions at rank <=3 β so if the definition is anywhere, it was already in
# round 0, and ranks 6-10 are where it is least likely to be. Raising EVIDENCE_K
# to 5 made this worse in both directions at once: escalation needs 6+ evidence
# chunks to fire at all, and the rounds it does reach are even further down.
#
# What would justify reviving it: a retry that changes the QUESTION (a different
# prompt, a larger tier) rather than one that changes the evidence to lower-ranked
# evidence. That is a different feature, not this constant.
MAX_ESCALATION_ROUNDS = 0
CONFLICT_OVERLAP_THRESHOLD = 0.4
DUPLICATE_OVERLAP_THRESHOLD = 0.8
@lru_cache(maxsize=4)
def load_yaml(name: str) -> dict:
with open(CONFIG_DIR / name, encoding="utf-8") as fh:
return yaml.safe_load(fh)
def labels_for(variant: str = LABELS_VARIANT) -> tuple[list[str], float]:
"""Returns (labels, threshold) for a label variant."""
cfg = load_yaml("labels.yaml")
labels = cfg.get(variant) or cfg.get(LABELS_VARIANT) or []
return list(labels), float(cfg.get("threshold", SPAN_SCORE_THRESHOLD))
|