ishaq101's picture
/fix parsing and term extract (#21)
f07443e
Raw History Blame Contribute Delete
17.4 kB
"""Tunables for the knowledge-extraction pipeline.
Every value here was calibrated on real documents and each one has a reason
recorded in KNOWLEDGE_PIPELINE_CALIBRATION.md. Change them deliberately β€” most
were arrived at by a measurement, and two of them (`FUZZY_MIN_LEN`,
`CACHE_MIN_TOKENS`) fix bugs that are silent when reintroduced.
Label and cue sets live in `config/*.yaml` so they can be tuned without a code
change: label phrasing is the main recall lever and the filter is very sensitive
to it.
"""
from __future__ import annotations
from functools import lru_cache
from pathlib import Path
import yaml
CONFIG_DIR = Path(__file__).resolve().parent / "config"
# ── Term filter ─────────────────────────────────────────────────────────
# Variant C beat both the English-default and Indonesian-phrasing label sets:
# the other two missed the same class (mining activities and materials).
LABELS_VARIANT = "broad"
# 0.25, not 0.35: measured recall 0.854 @ 0.25 vs 0.658 @ 0.35 on `broad`.
# Precision falls (0.41 vs 0.50) and that is the intended trade β€” the filter is
# deliberately over-inclusive, clustering and ranking absorb the noise, and a
# term the filter never proposes can never be recovered downstream.
SPAN_SCORE_THRESHOLD = 0.25
# The span NER model truncates past ~384 of its own tokens and *warns rather
# than failing*, so a long chunk silently loses its tail. Indonesian technical
# prose subword-tokenises at roughly 2.5x, so 220-word windows still tripped the
# cap; 130 did not. Chunks are fed as overlapping windows with offsets remapped.
#
# ⚠️ STAYS AT 130. A 2026-09-10 sweep appeared to show 256 was free β€” recall flat
# across 130/180/256/320 (E1 0.878 at every size) while clusters fell 6%/23%/34%
# across the three test documents. It was raised, then REVERTED the same day after
# counting the truncation directly with GLiNER's own tokeniser:
#
# windows exceeding the 384-token cap, by WINDOW_WORDS
# 130 180 200 256 320
# BUMA (ID) 1 1 1 1 1 <- pre-existing, see below
# Komatsu (EN) 0 0 0 1 10
# Open Pit (EN) 0 4 11 46 49 <- 46 of 99 windows!
#
# At 256 nearly HALF of the Open Pit textbook's windows lose their tail. The
# "flat recall" that justified the raise was measured on BUMA, the one document
# whose truncation count does not change with window size β€” so the metric was
# structurally blind to the cost, in exactly the way E1 was blind to the
# clustering over-merge. The cluster reduction was not better grouping; it was
# content disappearing.
#
# This vindicates the original reasoning rather than overturning it: the ~2.5x
# Indonesian subword ratio was the motivating case, but dense ENGLISH technical
# prose also trips the cap by 180. 130 is the only size measured safe on both
# English documents.
#
# ⚠️ Separate, PRE-EXISTING finding from the same measurement: BUMA has one window
# that exceeds the cap at EVERY size including 130 (409 tokens from 130 words β€” a
# dense Indonesian chunk). One chunk in that document has always been losing its
# tail. Not caused by this work and not fixed by it; a smaller window or a
# per-chunk token budget would be the fix, and neither is measured yet.
WINDOW_WORDS = 130
WINDOW_OVERLAP = 30
SPAN_TOKEN_CAP = 12
# ── The per-window SUBWORD budget (fixes the pre-existing over-cap window) ──
#
# `WINDOW_WORDS` counts WORDS; the model truncates on SUBWORDS, and the ratio
# between them is a property of the text, not of the setting. Measured with
# GLiNER's own tokeniser across all three documents, 2026-09-11:
#
# windows over the 384-token cap chars/subword (min / mean)
# BUMA 31 1 1.57 / 3.43
# Komatsu 144 0 2.53 / 3.58
# Open Pit 193 0 2.79 / 4.10
#
# One window in 368 is over, and it is the one the 2026-09-10 sweep found and
# left unfixed: `STD_2026_006_MNO::0024`, 130 words and 738 characters, which
# tokenise to 409 subwords. That chunk has ALWAYS been losing its tail, at every
# window size including 130, and the model warns rather than failing so nothing
# ever surfaced it.
#
# The fix measures instead of guessing: each window is checked against the
# model's own tokeniser and split only if it does not fit. A character budget
# would have to assume the worst ratio seen anywhere (1.57) and would therefore
# split at ~600 characters β€” chopping up the other 367 windows, which fit
# comfortably, to fix one.
SPAN_SUBWORD_CAP = 384
# Split at 90% of the cap. The tokeniser count excludes the special tokens and
# the label prompt GLiNER prepends, so measuring exactly to 384 still truncates.
SPAN_SUBWORD_MARGIN = 0.9
# Used only when the tokeniser cannot be located on the model (a GLiNER version
# change). 1.57 is the tightest ratio measured anywhere in the corpus, so this
# over-splits rather than silently truncating β€” the safe direction.
SPAN_FALLBACK_CHARS_PER_TOKEN = 1.57
# A split halves and recurses. 4 takes a 130-word window down to ~8 words, which
# is past any plausible cap; the bound exists so a pathological input cannot
# recurse indefinitely.
SPAN_MAX_SPLIT_DEPTH = 4
# Flat (non-overlapping) span decoding. MEASURED AND KEPT 2026-09-10: nested
# decoding (`flat_ner=False`) was tested on all three documents on the hypothesis
# that flat mode was discarding the parent/child spans compound terms need. It
# was not. Nested added 7-12% more mentions, delivered EXACTLY ZERO recall gain
# (E1 0.878 either way, E1c identical too), and increased token-subset
# over-merges by 40-80% (BUMA 15->24, Komatsu 18->32, Open Pit 97->134).
# `span_filter` leaves this at the library default rather than passing it; this
# constant documents the decision. Do not re-open without a new measurement.
SPAN_FLAT_NER = True
# ── Clustering ──────────────────────────────────────────────────────────
# ⚠️ BOTH CONSTANTS BELOW ARE RETIRED as of 2026-09-10. The fuzzy matching pass
# they configured is commented out at its call site in `cluster/cluster.py`; they
# are still imported there because the `_fuzzy_match` function itself is parked
# (comment-out-don't-delete), and it reads them. Changing them now affects
# NOTHING β€” the pass does not run.
#
# Why it was retired, measured across three documents: `token_set_ratio` returns
# exactly 100 when one token set is a subset of the other, so `FUZZY_THRESHOLD`
# was inert and head nouns always absorbed their compounds. 131 of 140 merges
# were subset matches (all wrong); of the 9 remaining, 6 were plain plurals β€”
# now handled deterministically by `normalize(lemma=True)` β€” and 3 were distinct
# metrics wrongly merged. Retiring the pass took subset-merges to 0 on all three
# documents and lifted post-cluster gold recall 0.8293 -> 0.8537.
#
# `FUZZY_MIN_LEN` did its own job correctly to the end: it stopped "PA" and "UA"
# colliding. It simply never gated multi-word English compounds, which is the
# case that did the damage.
FUZZY_THRESHOLD = 92
FUZZY_MIN_LEN = 5
# ── Evidence ranking ────────────────────────────────────────────────────
# 3 -> 5 on 2026-09-10 (Rifqi's call). How many top-ranked chunks each extraction
# call sees. Three things this changes, recorded because none is obvious:
#
# 1. **Cost.** Evidence is the variable part of the prompt, so this is the single
# biggest lever on spend per call after cluster count. Measured effect on the
# three test documents: see CHANGELOG_v2.md. It also lowers the CACHE HIT RATE,
# because the fixed cached prefix becomes a smaller share of each prompt.
#
# 2. **Escalation gets rarer but deeper.** `rounds_available` needs more than K
# evidence chunks to fire at all, so at 5 a cluster needs 6+ (not 4+) to
# escalate; when it does, a round covers ranks 6-10 instead of 4-6. On the three
# test documents most clusters have 1-2 evidence chunks, so escalation fires
# less often than before.
#
# 3. **It is NOT supported by the ranker measurement, and that is fine.** The
# 2026-09-10 ceiling check found 16 of 16 gold definitions already at rank <=3
# (14 at rank 1), so on the BUMA standard ranks 4-5 add cost and no recall. The
# case for 5 is the two ENGLISH documents, where definitions are dispersed
# across pages and there is no gold set to measure with β€” a judgement that more
# context beats a cheaper call on that document class. Revisit with a number if
# a second gold set ever exists.
EVIDENCE_K = 5
CUE_PROXIMITY_CHARS = 100
# How many LINKED partner chunks may be pulled into an evidence window on top of
# the ranked K (0.4.0 two-way table pointers; added 2026-09-11). 1, not more.
#
# A table keeps its own chunk, and the prose that introduces it is a separate
# chunk; `referenced_by` records the pair in both directions. Measured on the
# three v2 artifacts, ranking only (free β€” no GLiNER, no spend):
#
# clusters whose top-5 touches a linked pair, by side
# clusters touches a pair BOTH sides only ONE side
# BUMA 83 31 0 31
# Komatsu 93 55 0 55
# Open Pit 427 70 0 70
#
# **Zero clusters see both sides.** The ranker never co-selects a table and the
# prose that names it, so the partner is genuinely unseen content rather than a
# duplicate β€” and that prose is frequently the table's only title ("Troubleshooting
# of engine (S-mode)" above a bare `S-1 | 6` grid).
#
# Cost AT THE CAP OF 1, measured on the same artifacts: evidence the model reads
# +14.5% / +24.5% / +9.9%, which is +3.4k / +11.2k / +59.5k tokens, i.e.
# **+1.6% / +4.1% / +4.2%** of each document's total v2 prompt spend (the fixed
# cached prefix dominates). Open Pit moves from 71% to ~74% of the 2M ingest
# ceiling. Clusters that actually gain a partner: 31/83, 55/93, 70/427.
#
# **Expansion happens in `top_k`, NOT in `rank_evidence`.** Two invariants depend
# on that placement:
#
# 1. The full ranked list stays untouched, so nothing that reads
# `evidence_chunk_ids` β€” `rounds_available` included β€” sees a different number.
# Appending partners to the RANKED list would inflate the evidence count and
# change escalation's arithmetic as a side effect. That is currently moot
# (`MAX_ESCALATION_ROUNDS = 0` since 2026-09-11; escalation produced 0
# definitions in 48 calls) but the separation is what keeps this change
# orthogonal to that one, and to whatever replaces it.
# 2. `evidence_text` (the span check) and `evidence_block` (the prompt) both take
# their ids from `top_k`, so expanding there keeps "what the model reads is
# exactly what the span check searches" true for free. Expanding in the prompt
# renderer instead would break it, and every quote from a partner chunk would
# be nulled as unlocatable. Verified: 0 parity failures across all 603 clusters.
#
# Set to 0 to disable. Two honest scope notes:
#
# - The document this helps most (Komatsu, 59% of clusters gain a partner) is the
# one that returns 0 definitions for GENRE reasons. This is groundwork for the
# table branch more than a glossary win.
# - The counts above were measured on the v2 artifacts and the RECORDED v2 filter
# payloads, so they predate the subword window-splitting change. Mention and
# cluster counts move with that change; the pairing structure does not.
EVIDENCE_LINK_PARTNERS = 1
EVIDENCE_WEIGHTS: dict[str, float] = {
"definitional_cue_near": 5.0,
"term_in_heading": 4.0,
"in_legend_block": 3.5,
# A figure whose CAPTION names this term sits in this chunk. Built and
# measured 2026-09-11, then set to **0.0 β€” OFF, because it measured as a
# no-op.** The code stays because the A/B harness and the reasoning are worth
# more than the three lines, and a diagram-heavy document could revive it.
#
# It was built to explain `definitions_citing_a_caption = 0` on all three
# documents in the first paid v2 run. Two hypotheses, both now falsified:
#
# 1. "Captions don't reach the term filter." FALSE β€” a caption is part of
# `Chunk.text`, so GLiNER already scans it and already produces mentions
# from it (2 of 2 captions on BUMA, 3 of 8 on Open Pit).
# 2. "Figure-bearing chunks don't get selected." FALSE β€” a captioned chunk is
# already inside the top-5 for 22 of 83 BUMA clusters and 64 of 427 Open
# Pit clusters. Turning this weight up to 2.5 changed the top-5 of **1
# cluster out of 603** across all three documents, and moved zero clusters
# from "no figure" to "has a figure".
#
# The actual reason is simpler, and visible by reading them: **not one caption
# in the corpus is a definition.** They are figure titles β€” "Figure 1.3.
# Relative ability to influence costs (Lee, 1984)", "Gambar 2.1 Analisis
# Gain/Loss Berdasarkan Setiap Parameter Produksi". A model declining to quote
# one as a definition is CORRECT, not a retrieval failure.
#
# If revived, weight it BELOW `term_in_heading`: a heading naming a term means
# "this section is about X"; a caption naming it means only "a figure about X
# is here".
"asset_caption_names_term": 0.0,
"formula_present": 2.0,
"bold_or_italic": 1.5,
"first_occurrence": 1.0,
"tabular_penalty": -3.0,
}
# ── Chunking ────────────────────────────────────────────────────────────
MAX_CHUNK_TOKENS = 1500
MAX_HEADING_LEN = 90
BOILERPLATE_MIN_FRAC = 0.6
# ── Extraction ──────────────────────────────────────────────────────────
TEMPERATURE = 0.0
# OpenAI-family prompt caching does not engage AT ALL below this many prompt
# tokens, so a shorter fixed prefix caches nothing and costs ~10x on input. The
# measured hit rate at/above it was 54%.
CACHE_MIN_TOKENS = 1024
# ── Validation ──────────────────────────────────────────────────────────
# ⚠️ RETIRED 2026-09-11 by setting it to 0 β€” the loop and its call site are
# untouched (comment-out-don't-delete), and restoring it is this one character.
#
# Finding 6 asked for a number either way. Here it is, counted off the six
# persisted paid runs (v1 K=3 and v2 K=5, three documents):
#
# glossary calls clusters extra calls definitions FROM escalation
# BUMA v1 69 66 3 0
# BUMA v2 83 83 0 0
# Komatsu v1 94 75 19 0
# Komatsu v2 102 93 9 0
# Open Pit v1 361 349 12 0
# Open Pit v2 430 425 5 0
#
# 48 escalation calls across three documents and two configurations, and NOT ONE
# of them produced a definition the first round had missed. `extraction_status`
# is set to "escalated" only when a later round succeeds where round 0 failed,
# and it appears zero times in every one of those six files.
#
# Why it fails is worth keeping, because it argues against reviving it naively:
# escalation hands the model the NEXT k evidence chunks, ranked below the ones it
# already declined. The 2026-09-10 ranker ceiling check found 16 of 18 gold
# definitions at rank <=3 β€” so if the definition is anywhere, it was already in
# round 0, and ranks 6-10 are where it is least likely to be. Raising EVIDENCE_K
# to 5 made this worse in both directions at once: escalation needs 6+ evidence
# chunks to fire at all, and the rounds it does reach are even further down.
#
# What would justify reviving it: a retry that changes the QUESTION (a different
# prompt, a larger tier) rather than one that changes the evidence to lower-ranked
# evidence. That is a different feature, not this constant.
MAX_ESCALATION_ROUNDS = 0
CONFLICT_OVERLAP_THRESHOLD = 0.4
DUPLICATE_OVERLAP_THRESHOLD = 0.8
@lru_cache(maxsize=4)
def load_yaml(name: str) -> dict:
with open(CONFIG_DIR / name, encoding="utf-8") as fh:
return yaml.safe_load(fh)
def labels_for(variant: str = LABELS_VARIANT) -> tuple[list[str], float]:
"""Returns (labels, threshold) for a label variant."""
cfg = load_yaml("labels.yaml")
labels = cfg.get(variant) or cfg.get(LABELS_VARIANT) or []
return list(labels), float(cfg.get("threshold", SPAN_SCORE_THRESHOLD))