Download src/knowledge_extraction/settings.py from DataEyond/Agentic-Service-Data-Eyond-Catalog: direct link, hf CLI and curl.
- Browser
- Download file 17.4 kB
-
https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_extraction/settings.py
- Command line
-
hf download hf://spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/src/knowledge_extraction/settings.py
-
curl -L -o settings.py https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_extraction/settings.py
17.4 kB
| """Tunables for the knowledge-extraction pipeline. | |
| Every value here was calibrated on real documents and each one has a reason | |
| recorded in KNOWLEDGE_PIPELINE_CALIBRATION.md. Change them deliberately β most | |
| were arrived at by a measurement, and two of them (`FUZZY_MIN_LEN`, | |
| `CACHE_MIN_TOKENS`) fix bugs that are silent when reintroduced. | |
| Label and cue sets live in `config/*.yaml` so they can be tuned without a code | |
| change: label phrasing is the main recall lever and the filter is very sensitive | |
| to it. | |
| """ | |
| from __future__ import annotations | |
| from functools import lru_cache | |
| from pathlib import Path | |
| import yaml | |
| CONFIG_DIR = Path(__file__).resolve().parent / "config" | |
| # ββ Term filter βββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # Variant C beat both the English-default and Indonesian-phrasing label sets: | |
| # the other two missed the same class (mining activities and materials). | |
| LABELS_VARIANT = "broad" | |
| # 0.25, not 0.35: measured recall 0.854 @ 0.25 vs 0.658 @ 0.35 on `broad`. | |
| # Precision falls (0.41 vs 0.50) and that is the intended trade β the filter is | |
| # deliberately over-inclusive, clustering and ranking absorb the noise, and a | |
| # term the filter never proposes can never be recovered downstream. | |
| SPAN_SCORE_THRESHOLD = 0.25 | |
| # The span NER model truncates past ~384 of its own tokens and *warns rather | |
| # than failing*, so a long chunk silently loses its tail. Indonesian technical | |
| # prose subword-tokenises at roughly 2.5x, so 220-word windows still tripped the | |
| # cap; 130 did not. Chunks are fed as overlapping windows with offsets remapped. | |
| # | |
| # β οΈ STAYS AT 130. A 2026-09-10 sweep appeared to show 256 was free β recall flat | |
| # across 130/180/256/320 (E1 0.878 at every size) while clusters fell 6%/23%/34% | |
| # across the three test documents. It was raised, then REVERTED the same day after | |
| # counting the truncation directly with GLiNER's own tokeniser: | |
| # | |
| # windows exceeding the 384-token cap, by WINDOW_WORDS | |
| # 130 180 200 256 320 | |
| # BUMA (ID) 1 1 1 1 1 <- pre-existing, see below | |
| # Komatsu (EN) 0 0 0 1 10 | |
| # Open Pit (EN) 0 4 11 46 49 <- 46 of 99 windows! | |
| # | |
| # At 256 nearly HALF of the Open Pit textbook's windows lose their tail. The | |
| # "flat recall" that justified the raise was measured on BUMA, the one document | |
| # whose truncation count does not change with window size β so the metric was | |
| # structurally blind to the cost, in exactly the way E1 was blind to the | |
| # clustering over-merge. The cluster reduction was not better grouping; it was | |
| # content disappearing. | |
| # | |
| # This vindicates the original reasoning rather than overturning it: the ~2.5x | |
| # Indonesian subword ratio was the motivating case, but dense ENGLISH technical | |
| # prose also trips the cap by 180. 130 is the only size measured safe on both | |
| # English documents. | |
| # | |
| # β οΈ Separate, PRE-EXISTING finding from the same measurement: BUMA has one window | |
| # that exceeds the cap at EVERY size including 130 (409 tokens from 130 words β a | |
| # dense Indonesian chunk). One chunk in that document has always been losing its | |
| # tail. Not caused by this work and not fixed by it; a smaller window or a | |
| # per-chunk token budget would be the fix, and neither is measured yet. | |
| WINDOW_WORDS = 130 | |
| WINDOW_OVERLAP = 30 | |
| SPAN_TOKEN_CAP = 12 | |
| # ββ The per-window SUBWORD budget (fixes the pre-existing over-cap window) ββ | |
| # | |
| # `WINDOW_WORDS` counts WORDS; the model truncates on SUBWORDS, and the ratio | |
| # between them is a property of the text, not of the setting. Measured with | |
| # GLiNER's own tokeniser across all three documents, 2026-09-11: | |
| # | |
| # windows over the 384-token cap chars/subword (min / mean) | |
| # BUMA 31 1 1.57 / 3.43 | |
| # Komatsu 144 0 2.53 / 3.58 | |
| # Open Pit 193 0 2.79 / 4.10 | |
| # | |
| # One window in 368 is over, and it is the one the 2026-09-10 sweep found and | |
| # left unfixed: `STD_2026_006_MNO::0024`, 130 words and 738 characters, which | |
| # tokenise to 409 subwords. That chunk has ALWAYS been losing its tail, at every | |
| # window size including 130, and the model warns rather than failing so nothing | |
| # ever surfaced it. | |
| # | |
| # The fix measures instead of guessing: each window is checked against the | |
| # model's own tokeniser and split only if it does not fit. A character budget | |
| # would have to assume the worst ratio seen anywhere (1.57) and would therefore | |
| # split at ~600 characters β chopping up the other 367 windows, which fit | |
| # comfortably, to fix one. | |
| SPAN_SUBWORD_CAP = 384 | |
| # Split at 90% of the cap. The tokeniser count excludes the special tokens and | |
| # the label prompt GLiNER prepends, so measuring exactly to 384 still truncates. | |
| SPAN_SUBWORD_MARGIN = 0.9 | |
| # Used only when the tokeniser cannot be located on the model (a GLiNER version | |
| # change). 1.57 is the tightest ratio measured anywhere in the corpus, so this | |
| # over-splits rather than silently truncating β the safe direction. | |
| SPAN_FALLBACK_CHARS_PER_TOKEN = 1.57 | |
| # A split halves and recurses. 4 takes a 130-word window down to ~8 words, which | |
| # is past any plausible cap; the bound exists so a pathological input cannot | |
| # recurse indefinitely. | |
| SPAN_MAX_SPLIT_DEPTH = 4 | |
| # Flat (non-overlapping) span decoding. MEASURED AND KEPT 2026-09-10: nested | |
| # decoding (`flat_ner=False`) was tested on all three documents on the hypothesis | |
| # that flat mode was discarding the parent/child spans compound terms need. It | |
| # was not. Nested added 7-12% more mentions, delivered EXACTLY ZERO recall gain | |
| # (E1 0.878 either way, E1c identical too), and increased token-subset | |
| # over-merges by 40-80% (BUMA 15->24, Komatsu 18->32, Open Pit 97->134). | |
| # `span_filter` leaves this at the library default rather than passing it; this | |
| # constant documents the decision. Do not re-open without a new measurement. | |
| SPAN_FLAT_NER = True | |
| # ββ Clustering ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # β οΈ BOTH CONSTANTS BELOW ARE RETIRED as of 2026-09-10. The fuzzy matching pass | |
| # they configured is commented out at its call site in `cluster/cluster.py`; they | |
| # are still imported there because the `_fuzzy_match` function itself is parked | |
| # (comment-out-don't-delete), and it reads them. Changing them now affects | |
| # NOTHING β the pass does not run. | |
| # | |
| # Why it was retired, measured across three documents: `token_set_ratio` returns | |
| # exactly 100 when one token set is a subset of the other, so `FUZZY_THRESHOLD` | |
| # was inert and head nouns always absorbed their compounds. 131 of 140 merges | |
| # were subset matches (all wrong); of the 9 remaining, 6 were plain plurals β | |
| # now handled deterministically by `normalize(lemma=True)` β and 3 were distinct | |
| # metrics wrongly merged. Retiring the pass took subset-merges to 0 on all three | |
| # documents and lifted post-cluster gold recall 0.8293 -> 0.8537. | |
| # | |
| # `FUZZY_MIN_LEN` did its own job correctly to the end: it stopped "PA" and "UA" | |
| # colliding. It simply never gated multi-word English compounds, which is the | |
| # case that did the damage. | |
| FUZZY_THRESHOLD = 92 | |
| FUZZY_MIN_LEN = 5 | |
| # ββ Evidence ranking ββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # 3 -> 5 on 2026-09-10 (Rifqi's call). How many top-ranked chunks each extraction | |
| # call sees. Three things this changes, recorded because none is obvious: | |
| # | |
| # 1. **Cost.** Evidence is the variable part of the prompt, so this is the single | |
| # biggest lever on spend per call after cluster count. Measured effect on the | |
| # three test documents: see CHANGELOG_v2.md. It also lowers the CACHE HIT RATE, | |
| # because the fixed cached prefix becomes a smaller share of each prompt. | |
| # | |
| # 2. **Escalation gets rarer but deeper.** `rounds_available` needs more than K | |
| # evidence chunks to fire at all, so at 5 a cluster needs 6+ (not 4+) to | |
| # escalate; when it does, a round covers ranks 6-10 instead of 4-6. On the three | |
| # test documents most clusters have 1-2 evidence chunks, so escalation fires | |
| # less often than before. | |
| # | |
| # 3. **It is NOT supported by the ranker measurement, and that is fine.** The | |
| # 2026-09-10 ceiling check found 16 of 16 gold definitions already at rank <=3 | |
| # (14 at rank 1), so on the BUMA standard ranks 4-5 add cost and no recall. The | |
| # case for 5 is the two ENGLISH documents, where definitions are dispersed | |
| # across pages and there is no gold set to measure with β a judgement that more | |
| # context beats a cheaper call on that document class. Revisit with a number if | |
| # a second gold set ever exists. | |
| EVIDENCE_K = 5 | |
| CUE_PROXIMITY_CHARS = 100 | |
| # How many LINKED partner chunks may be pulled into an evidence window on top of | |
| # the ranked K (0.4.0 two-way table pointers; added 2026-09-11). 1, not more. | |
| # | |
| # A table keeps its own chunk, and the prose that introduces it is a separate | |
| # chunk; `referenced_by` records the pair in both directions. Measured on the | |
| # three v2 artifacts, ranking only (free β no GLiNER, no spend): | |
| # | |
| # clusters whose top-5 touches a linked pair, by side | |
| # clusters touches a pair BOTH sides only ONE side | |
| # BUMA 83 31 0 31 | |
| # Komatsu 93 55 0 55 | |
| # Open Pit 427 70 0 70 | |
| # | |
| # **Zero clusters see both sides.** The ranker never co-selects a table and the | |
| # prose that names it, so the partner is genuinely unseen content rather than a | |
| # duplicate β and that prose is frequently the table's only title ("Troubleshooting | |
| # of engine (S-mode)" above a bare `S-1 | 6` grid). | |
| # | |
| # Cost AT THE CAP OF 1, measured on the same artifacts: evidence the model reads | |
| # +14.5% / +24.5% / +9.9%, which is +3.4k / +11.2k / +59.5k tokens, i.e. | |
| # **+1.6% / +4.1% / +4.2%** of each document's total v2 prompt spend (the fixed | |
| # cached prefix dominates). Open Pit moves from 71% to ~74% of the 2M ingest | |
| # ceiling. Clusters that actually gain a partner: 31/83, 55/93, 70/427. | |
| # | |
| # **Expansion happens in `top_k`, NOT in `rank_evidence`.** Two invariants depend | |
| # on that placement: | |
| # | |
| # 1. The full ranked list stays untouched, so nothing that reads | |
| # `evidence_chunk_ids` β `rounds_available` included β sees a different number. | |
| # Appending partners to the RANKED list would inflate the evidence count and | |
| # change escalation's arithmetic as a side effect. That is currently moot | |
| # (`MAX_ESCALATION_ROUNDS = 0` since 2026-09-11; escalation produced 0 | |
| # definitions in 48 calls) but the separation is what keeps this change | |
| # orthogonal to that one, and to whatever replaces it. | |
| # 2. `evidence_text` (the span check) and `evidence_block` (the prompt) both take | |
| # their ids from `top_k`, so expanding there keeps "what the model reads is | |
| # exactly what the span check searches" true for free. Expanding in the prompt | |
| # renderer instead would break it, and every quote from a partner chunk would | |
| # be nulled as unlocatable. Verified: 0 parity failures across all 603 clusters. | |
| # | |
| # Set to 0 to disable. Two honest scope notes: | |
| # | |
| # - The document this helps most (Komatsu, 59% of clusters gain a partner) is the | |
| # one that returns 0 definitions for GENRE reasons. This is groundwork for the | |
| # table branch more than a glossary win. | |
| # - The counts above were measured on the v2 artifacts and the RECORDED v2 filter | |
| # payloads, so they predate the subword window-splitting change. Mention and | |
| # cluster counts move with that change; the pairing structure does not. | |
| EVIDENCE_LINK_PARTNERS = 1 | |
| EVIDENCE_WEIGHTS: dict[str, float] = { | |
| "definitional_cue_near": 5.0, | |
| "term_in_heading": 4.0, | |
| "in_legend_block": 3.5, | |
| # A figure whose CAPTION names this term sits in this chunk. Built and | |
| # measured 2026-09-11, then set to **0.0 β OFF, because it measured as a | |
| # no-op.** The code stays because the A/B harness and the reasoning are worth | |
| # more than the three lines, and a diagram-heavy document could revive it. | |
| # | |
| # It was built to explain `definitions_citing_a_caption = 0` on all three | |
| # documents in the first paid v2 run. Two hypotheses, both now falsified: | |
| # | |
| # 1. "Captions don't reach the term filter." FALSE β a caption is part of | |
| # `Chunk.text`, so GLiNER already scans it and already produces mentions | |
| # from it (2 of 2 captions on BUMA, 3 of 8 on Open Pit). | |
| # 2. "Figure-bearing chunks don't get selected." FALSE β a captioned chunk is | |
| # already inside the top-5 for 22 of 83 BUMA clusters and 64 of 427 Open | |
| # Pit clusters. Turning this weight up to 2.5 changed the top-5 of **1 | |
| # cluster out of 603** across all three documents, and moved zero clusters | |
| # from "no figure" to "has a figure". | |
| # | |
| # The actual reason is simpler, and visible by reading them: **not one caption | |
| # in the corpus is a definition.** They are figure titles β "Figure 1.3. | |
| # Relative ability to influence costs (Lee, 1984)", "Gambar 2.1 Analisis | |
| # Gain/Loss Berdasarkan Setiap Parameter Produksi". A model declining to quote | |
| # one as a definition is CORRECT, not a retrieval failure. | |
| # | |
| # If revived, weight it BELOW `term_in_heading`: a heading naming a term means | |
| # "this section is about X"; a caption naming it means only "a figure about X | |
| # is here". | |
| "asset_caption_names_term": 0.0, | |
| "formula_present": 2.0, | |
| "bold_or_italic": 1.5, | |
| "first_occurrence": 1.0, | |
| "tabular_penalty": -3.0, | |
| } | |
| # ββ Chunking ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| MAX_CHUNK_TOKENS = 1500 | |
| MAX_HEADING_LEN = 90 | |
| BOILERPLATE_MIN_FRAC = 0.6 | |
| # ββ Extraction ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| TEMPERATURE = 0.0 | |
| # OpenAI-family prompt caching does not engage AT ALL below this many prompt | |
| # tokens, so a shorter fixed prefix caches nothing and costs ~10x on input. The | |
| # measured hit rate at/above it was 54%. | |
| CACHE_MIN_TOKENS = 1024 | |
| # ββ Validation ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # β οΈ RETIRED 2026-09-11 by setting it to 0 β the loop and its call site are | |
| # untouched (comment-out-don't-delete), and restoring it is this one character. | |
| # | |
| # Finding 6 asked for a number either way. Here it is, counted off the six | |
| # persisted paid runs (v1 K=3 and v2 K=5, three documents): | |
| # | |
| # glossary calls clusters extra calls definitions FROM escalation | |
| # BUMA v1 69 66 3 0 | |
| # BUMA v2 83 83 0 0 | |
| # Komatsu v1 94 75 19 0 | |
| # Komatsu v2 102 93 9 0 | |
| # Open Pit v1 361 349 12 0 | |
| # Open Pit v2 430 425 5 0 | |
| # | |
| # 48 escalation calls across three documents and two configurations, and NOT ONE | |
| # of them produced a definition the first round had missed. `extraction_status` | |
| # is set to "escalated" only when a later round succeeds where round 0 failed, | |
| # and it appears zero times in every one of those six files. | |
| # | |
| # Why it fails is worth keeping, because it argues against reviving it naively: | |
| # escalation hands the model the NEXT k evidence chunks, ranked below the ones it | |
| # already declined. The 2026-09-10 ranker ceiling check found 16 of 18 gold | |
| # definitions at rank <=3 β so if the definition is anywhere, it was already in | |
| # round 0, and ranks 6-10 are where it is least likely to be. Raising EVIDENCE_K | |
| # to 5 made this worse in both directions at once: escalation needs 6+ evidence | |
| # chunks to fire at all, and the rounds it does reach are even further down. | |
| # | |
| # What would justify reviving it: a retry that changes the QUESTION (a different | |
| # prompt, a larger tier) rather than one that changes the evidence to lower-ranked | |
| # evidence. That is a different feature, not this constant. | |
| MAX_ESCALATION_ROUNDS = 0 | |
| CONFLICT_OVERLAP_THRESHOLD = 0.4 | |
| DUPLICATE_OVERLAP_THRESHOLD = 0.8 | |
| def load_yaml(name: str) -> dict: | |
| with open(CONFIG_DIR / name, encoding="utf-8") as fh: | |
| return yaml.safe_load(fh) | |
| def labels_for(variant: str = LABELS_VARIANT) -> tuple[list[str], float]: | |
| """Returns (labels, threshold) for a label variant.""" | |
| cfg = load_yaml("labels.yaml") | |
| labels = cfg.get(variant) or cfg.get(LABELS_VARIANT) or [] | |
| return list(labels), float(cfg.get("threshold", SPAN_SCORE_THRESHOLD)) | |