File size: 17,365 Bytes
b68816f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
f07443e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
b68816f
 
 
 
f07443e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
b68816f
f07443e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
b68816f
 
 
 
f07443e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
b68816f
 
f07443e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
b68816f
 
 
 
f07443e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
b68816f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
f07443e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
b68816f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
"""Tunables for the knowledge-extraction pipeline.

Every value here was calibrated on real documents and each one has a reason
recorded in KNOWLEDGE_PIPELINE_CALIBRATION.md. Change them deliberately β€” most
were arrived at by a measurement, and two of them (`FUZZY_MIN_LEN`,
`CACHE_MIN_TOKENS`) fix bugs that are silent when reintroduced.

Label and cue sets live in `config/*.yaml` so they can be tuned without a code
change: label phrasing is the main recall lever and the filter is very sensitive
to it.
"""

from __future__ import annotations

from functools import lru_cache
from pathlib import Path

import yaml

CONFIG_DIR = Path(__file__).resolve().parent / "config"

# ── Term filter ─────────────────────────────────────────────────────────
# Variant C beat both the English-default and Indonesian-phrasing label sets:
# the other two missed the same class (mining activities and materials).
LABELS_VARIANT = "broad"

# 0.25, not 0.35: measured recall 0.854 @ 0.25 vs 0.658 @ 0.35 on `broad`.
# Precision falls (0.41 vs 0.50) and that is the intended trade β€” the filter is
# deliberately over-inclusive, clustering and ranking absorb the noise, and a
# term the filter never proposes can never be recovered downstream.
SPAN_SCORE_THRESHOLD = 0.25

# The span NER model truncates past ~384 of its own tokens and *warns rather
# than failing*, so a long chunk silently loses its tail. Indonesian technical
# prose subword-tokenises at roughly 2.5x, so 220-word windows still tripped the
# cap; 130 did not. Chunks are fed as overlapping windows with offsets remapped.
#
# ⚠️ STAYS AT 130. A 2026-09-10 sweep appeared to show 256 was free β€” recall flat
# across 130/180/256/320 (E1 0.878 at every size) while clusters fell 6%/23%/34%
# across the three test documents. It was raised, then REVERTED the same day after
# counting the truncation directly with GLiNER's own tokeniser:
#
#   windows exceeding the 384-token cap, by WINDOW_WORDS
#                130    180    200    256    320
#   BUMA (ID)      1      1      1      1      1     <- pre-existing, see below
#   Komatsu (EN)   0      0      0      1     10
#   Open Pit (EN)  0      4     11     46     49     <- 46 of 99 windows!
#
# At 256 nearly HALF of the Open Pit textbook's windows lose their tail. The
# "flat recall" that justified the raise was measured on BUMA, the one document
# whose truncation count does not change with window size β€” so the metric was
# structurally blind to the cost, in exactly the way E1 was blind to the
# clustering over-merge. The cluster reduction was not better grouping; it was
# content disappearing.
#
# This vindicates the original reasoning rather than overturning it: the ~2.5x
# Indonesian subword ratio was the motivating case, but dense ENGLISH technical
# prose also trips the cap by 180. 130 is the only size measured safe on both
# English documents.
#
# ⚠️ Separate, PRE-EXISTING finding from the same measurement: BUMA has one window
# that exceeds the cap at EVERY size including 130 (409 tokens from 130 words β€” a
# dense Indonesian chunk). One chunk in that document has always been losing its
# tail. Not caused by this work and not fixed by it; a smaller window or a
# per-chunk token budget would be the fix, and neither is measured yet.
WINDOW_WORDS = 130
WINDOW_OVERLAP = 30
SPAN_TOKEN_CAP = 12

# ── The per-window SUBWORD budget (fixes the pre-existing over-cap window) ──
#
# `WINDOW_WORDS` counts WORDS; the model truncates on SUBWORDS, and the ratio
# between them is a property of the text, not of the setting. Measured with
# GLiNER's own tokeniser across all three documents, 2026-09-11:
#
#                windows   over the 384-token cap   chars/subword (min / mean)
#   BUMA              31                        1            1.57 / 3.43
#   Komatsu          144                        0            2.53 / 3.58
#   Open Pit         193                        0            2.79 / 4.10
#
# One window in 368 is over, and it is the one the 2026-09-10 sweep found and
# left unfixed: `STD_2026_006_MNO::0024`, 130 words and 738 characters, which
# tokenise to 409 subwords. That chunk has ALWAYS been losing its tail, at every
# window size including 130, and the model warns rather than failing so nothing
# ever surfaced it.
#
# The fix measures instead of guessing: each window is checked against the
# model's own tokeniser and split only if it does not fit. A character budget
# would have to assume the worst ratio seen anywhere (1.57) and would therefore
# split at ~600 characters β€” chopping up the other 367 windows, which fit
# comfortably, to fix one.
SPAN_SUBWORD_CAP = 384

# Split at 90% of the cap. The tokeniser count excludes the special tokens and
# the label prompt GLiNER prepends, so measuring exactly to 384 still truncates.
SPAN_SUBWORD_MARGIN = 0.9

# Used only when the tokeniser cannot be located on the model (a GLiNER version
# change). 1.57 is the tightest ratio measured anywhere in the corpus, so this
# over-splits rather than silently truncating β€” the safe direction.
SPAN_FALLBACK_CHARS_PER_TOKEN = 1.57

# A split halves and recurses. 4 takes a 130-word window down to ~8 words, which
# is past any plausible cap; the bound exists so a pathological input cannot
# recurse indefinitely.
SPAN_MAX_SPLIT_DEPTH = 4

# Flat (non-overlapping) span decoding. MEASURED AND KEPT 2026-09-10: nested
# decoding (`flat_ner=False`) was tested on all three documents on the hypothesis
# that flat mode was discarding the parent/child spans compound terms need. It
# was not. Nested added 7-12% more mentions, delivered EXACTLY ZERO recall gain
# (E1 0.878 either way, E1c identical too), and increased token-subset
# over-merges by 40-80% (BUMA 15->24, Komatsu 18->32, Open Pit 97->134).
# `span_filter` leaves this at the library default rather than passing it; this
# constant documents the decision. Do not re-open without a new measurement.
SPAN_FLAT_NER = True

# ── Clustering ──────────────────────────────────────────────────────────
# ⚠️ BOTH CONSTANTS BELOW ARE RETIRED as of 2026-09-10. The fuzzy matching pass
# they configured is commented out at its call site in `cluster/cluster.py`; they
# are still imported there because the `_fuzzy_match` function itself is parked
# (comment-out-don't-delete), and it reads them. Changing them now affects
# NOTHING β€” the pass does not run.
#
# Why it was retired, measured across three documents: `token_set_ratio` returns
# exactly 100 when one token set is a subset of the other, so `FUZZY_THRESHOLD`
# was inert and head nouns always absorbed their compounds. 131 of 140 merges
# were subset matches (all wrong); of the 9 remaining, 6 were plain plurals β€”
# now handled deterministically by `normalize(lemma=True)` β€” and 3 were distinct
# metrics wrongly merged. Retiring the pass took subset-merges to 0 on all three
# documents and lifted post-cluster gold recall 0.8293 -> 0.8537.
#
# `FUZZY_MIN_LEN` did its own job correctly to the end: it stopped "PA" and "UA"
# colliding. It simply never gated multi-word English compounds, which is the
# case that did the damage.
FUZZY_THRESHOLD = 92
FUZZY_MIN_LEN = 5

# ── Evidence ranking ────────────────────────────────────────────────────
# 3 -> 5 on 2026-09-10 (Rifqi's call). How many top-ranked chunks each extraction
# call sees. Three things this changes, recorded because none is obvious:
#
# 1. **Cost.** Evidence is the variable part of the prompt, so this is the single
#    biggest lever on spend per call after cluster count. Measured effect on the
#    three test documents: see CHANGELOG_v2.md. It also lowers the CACHE HIT RATE,
#    because the fixed cached prefix becomes a smaller share of each prompt.
#
# 2. **Escalation gets rarer but deeper.** `rounds_available` needs more than K
#    evidence chunks to fire at all, so at 5 a cluster needs 6+ (not 4+) to
#    escalate; when it does, a round covers ranks 6-10 instead of 4-6. On the three
#    test documents most clusters have 1-2 evidence chunks, so escalation fires
#    less often than before.
#
# 3. **It is NOT supported by the ranker measurement, and that is fine.** The
#    2026-09-10 ceiling check found 16 of 16 gold definitions already at rank <=3
#    (14 at rank 1), so on the BUMA standard ranks 4-5 add cost and no recall. The
#    case for 5 is the two ENGLISH documents, where definitions are dispersed
#    across pages and there is no gold set to measure with β€” a judgement that more
#    context beats a cheaper call on that document class. Revisit with a number if
#    a second gold set ever exists.
EVIDENCE_K = 5
CUE_PROXIMITY_CHARS = 100

# How many LINKED partner chunks may be pulled into an evidence window on top of
# the ranked K (0.4.0 two-way table pointers; added 2026-09-11). 1, not more.
#
# A table keeps its own chunk, and the prose that introduces it is a separate
# chunk; `referenced_by` records the pair in both directions. Measured on the
# three v2 artifacts, ranking only (free β€” no GLiNER, no spend):
#
#   clusters whose top-5 touches a linked pair, by side
#                 clusters   touches a pair   BOTH sides   only ONE side
#   BUMA              83           31              0             31
#   Komatsu           93           55              0             55
#   Open Pit         427           70              0             70
#
# **Zero clusters see both sides.** The ranker never co-selects a table and the
# prose that names it, so the partner is genuinely unseen content rather than a
# duplicate β€” and that prose is frequently the table's only title ("Troubleshooting
# of engine (S-mode)" above a bare `S-1 | 6` grid).
#
# Cost AT THE CAP OF 1, measured on the same artifacts: evidence the model reads
# +14.5% / +24.5% / +9.9%, which is +3.4k / +11.2k / +59.5k tokens, i.e.
# **+1.6% / +4.1% / +4.2%** of each document's total v2 prompt spend (the fixed
# cached prefix dominates). Open Pit moves from 71% to ~74% of the 2M ingest
# ceiling. Clusters that actually gain a partner: 31/83, 55/93, 70/427.
#
# **Expansion happens in `top_k`, NOT in `rank_evidence`.** Two invariants depend
# on that placement:
#
# 1. The full ranked list stays untouched, so nothing that reads
#    `evidence_chunk_ids` β€” `rounds_available` included β€” sees a different number.
#    Appending partners to the RANKED list would inflate the evidence count and
#    change escalation's arithmetic as a side effect. That is currently moot
#    (`MAX_ESCALATION_ROUNDS = 0` since 2026-09-11; escalation produced 0
#    definitions in 48 calls) but the separation is what keeps this change
#    orthogonal to that one, and to whatever replaces it.
# 2. `evidence_text` (the span check) and `evidence_block` (the prompt) both take
#    their ids from `top_k`, so expanding there keeps "what the model reads is
#    exactly what the span check searches" true for free. Expanding in the prompt
#    renderer instead would break it, and every quote from a partner chunk would
#    be nulled as unlocatable. Verified: 0 parity failures across all 603 clusters.
#
# Set to 0 to disable. Two honest scope notes:
#
# - The document this helps most (Komatsu, 59% of clusters gain a partner) is the
#   one that returns 0 definitions for GENRE reasons. This is groundwork for the
#   table branch more than a glossary win.
# - The counts above were measured on the v2 artifacts and the RECORDED v2 filter
#   payloads, so they predate the subword window-splitting change. Mention and
#   cluster counts move with that change; the pairing structure does not.
EVIDENCE_LINK_PARTNERS = 1

EVIDENCE_WEIGHTS: dict[str, float] = {
    "definitional_cue_near": 5.0,
    "term_in_heading": 4.0,
    "in_legend_block": 3.5,
    # A figure whose CAPTION names this term sits in this chunk. Built and
    # measured 2026-09-11, then set to **0.0 β€” OFF, because it measured as a
    # no-op.** The code stays because the A/B harness and the reasoning are worth
    # more than the three lines, and a diagram-heavy document could revive it.
    #
    # It was built to explain `definitions_citing_a_caption = 0` on all three
    # documents in the first paid v2 run. Two hypotheses, both now falsified:
    #
    # 1. "Captions don't reach the term filter." FALSE β€” a caption is part of
    #    `Chunk.text`, so GLiNER already scans it and already produces mentions
    #    from it (2 of 2 captions on BUMA, 3 of 8 on Open Pit).
    # 2. "Figure-bearing chunks don't get selected." FALSE β€” a captioned chunk is
    #    already inside the top-5 for 22 of 83 BUMA clusters and 64 of 427 Open
    #    Pit clusters. Turning this weight up to 2.5 changed the top-5 of **1
    #    cluster out of 603** across all three documents, and moved zero clusters
    #    from "no figure" to "has a figure".
    #
    # The actual reason is simpler, and visible by reading them: **not one caption
    # in the corpus is a definition.** They are figure titles β€” "Figure 1.3.
    # Relative ability to influence costs (Lee, 1984)", "Gambar 2.1 Analisis
    # Gain/Loss Berdasarkan Setiap Parameter Produksi". A model declining to quote
    # one as a definition is CORRECT, not a retrieval failure.
    #
    # If revived, weight it BELOW `term_in_heading`: a heading naming a term means
    # "this section is about X"; a caption naming it means only "a figure about X
    # is here".
    "asset_caption_names_term": 0.0,
    "formula_present": 2.0,
    "bold_or_italic": 1.5,
    "first_occurrence": 1.0,
    "tabular_penalty": -3.0,
}

# ── Chunking ────────────────────────────────────────────────────────────
MAX_CHUNK_TOKENS = 1500
MAX_HEADING_LEN = 90
BOILERPLATE_MIN_FRAC = 0.6

# ── Extraction ──────────────────────────────────────────────────────────
TEMPERATURE = 0.0

# OpenAI-family prompt caching does not engage AT ALL below this many prompt
# tokens, so a shorter fixed prefix caches nothing and costs ~10x on input. The
# measured hit rate at/above it was 54%.
CACHE_MIN_TOKENS = 1024

# ── Validation ──────────────────────────────────────────────────────────
# ⚠️ RETIRED 2026-09-11 by setting it to 0 β€” the loop and its call site are
# untouched (comment-out-don't-delete), and restoring it is this one character.
#
# Finding 6 asked for a number either way. Here it is, counted off the six
# persisted paid runs (v1 K=3 and v2 K=5, three documents):
#
#                    glossary calls   clusters   extra calls   definitions FROM escalation
#   BUMA      v1                 69         66             3                             0
#   BUMA      v2                 83         83             0                             0
#   Komatsu   v1                 94         75            19                             0
#   Komatsu   v2                102         93             9                             0
#   Open Pit  v1                361        349            12                             0
#   Open Pit  v2                430        425             5                             0
#
# 48 escalation calls across three documents and two configurations, and NOT ONE
# of them produced a definition the first round had missed. `extraction_status`
# is set to "escalated" only when a later round succeeds where round 0 failed,
# and it appears zero times in every one of those six files.
#
# Why it fails is worth keeping, because it argues against reviving it naively:
# escalation hands the model the NEXT k evidence chunks, ranked below the ones it
# already declined. The 2026-09-10 ranker ceiling check found 16 of 18 gold
# definitions at rank <=3 β€” so if the definition is anywhere, it was already in
# round 0, and ranks 6-10 are where it is least likely to be. Raising EVIDENCE_K
# to 5 made this worse in both directions at once: escalation needs 6+ evidence
# chunks to fire at all, and the rounds it does reach are even further down.
#
# What would justify reviving it: a retry that changes the QUESTION (a different
# prompt, a larger tier) rather than one that changes the evidence to lower-ranked
# evidence. That is a different feature, not this constant.
MAX_ESCALATION_ROUNDS = 0
CONFLICT_OVERLAP_THRESHOLD = 0.4
DUPLICATE_OVERLAP_THRESHOLD = 0.8


@lru_cache(maxsize=4)
def load_yaml(name: str) -> dict:
    with open(CONFIG_DIR / name, encoding="utf-8") as fh:
        return yaml.safe_load(fh)


def labels_for(variant: str = LABELS_VARIANT) -> tuple[list[str], float]:
    """Returns (labels, threshold) for a label variant."""
    cfg = load_yaml("labels.yaml")
    labels = cfg.get(variant) or cfg.get(LABELS_VARIANT) or []
    return list(labels), float(cfg.get("threshold", SPAN_SCORE_THRESHOLD))