Download src/knowledge_extraction/link.py from DataEyond/Agentic-Service-Data-Eyond-Catalog: direct link, hf CLI and curl.
- Browser
- Download file 9.23 kB
-
https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_extraction/link.py
- Command line
-
hf download hf://spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/src/knowledge_extraction/link.py
-
curl -L -o link.py https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_extraction/link.py
9.23 kB
| """Resolve references between the four artifacts. Deterministic, no LLM, no spend. | |
| Runs after every branch has produced entries, because a link can only be made | |
| once both of its ends exist. `GlossaryEntry` is the hub: formulas and rules | |
| point at terms, and a term points back at the formula that defines it. | |
| RuleEntry ──term_ids[]────► GlossaryEntry ◄──variables[].term_id── FormulaEntry | |
| └────formula_ids[]────────────────────────────────────────────────► ▲ | |
| GlossaryEntry ──defining_formula_id──────────────────┘ | |
| Three properties this stage must keep: | |
| 1. **A dangling link is null or an empty list — never a guess.** Same rule as | |
| span validation: an unresolvable reference is reported, not repaired. Linking | |
| is the one stage that could quietly invent structure, so it does not. | |
| 2. **Nothing here is model-supplied.** Everything is string matching over values | |
| already extracted and already span-checked, which is why no link can be | |
| hallucinated. | |
| 3. **The "appears in" edge is NOT stored.** That PA occurs inside | |
| `Production = MOHH x Qty x PA x UA x Pty` is recoverable by scanning | |
| `formulas[].variables[]`. Do not add a field for it — a derived edge that is | |
| also stored is an edge that can disagree with itself. | |
| Matching is word-boundary, never substring — the same trap the ranker and the | |
| CLI's legend stand-in both document. "PA" occurs inside "parameter", "pada", | |
| "capacity" and "composite"; substring matching produced 126 spurious PA mentions | |
| on a 9-page document. | |
| """ | |
| from __future__ import annotations | |
| from .cluster.normalize import normalize | |
| from .models import ClusterResult | |
| from .rank.evidence import _word_match | |
| def _index_terms(entries: list[dict], clustered: ClusterResult) -> dict[str, str]: | |
| """Normalised surface -> term_id, over canonical forms and every variant. | |
| Variants matter: a rule naming "Physical Availability" must reach the term | |
| stored as "PA", and only the cluster knows they are the same. | |
| """ | |
| by_canonical = {normalize(c.canonical): c for c in clustered.clusters} | |
| index: dict[str, str] = {} | |
| for entry in entries: | |
| term_id = entry.get("term_id") | |
| if not term_id: | |
| continue | |
| surfaces = [entry.get("term") or "", entry.get("full_name") or ""] | |
| cluster = by_canonical.get(normalize(entry.get("term") or "")) | |
| if cluster is not None: | |
| surfaces.extend(cluster.variants) | |
| for surface in surfaces: | |
| key = normalize(surface) | |
| # First writer wins: entries arrive frequency-ordered, so an | |
| # ambiguous surface resolves to the more frequent term rather than | |
| # to whichever happened to be processed last. | |
| if key and key not in index: | |
| index[key] = term_id | |
| return index | |
| def _resolve(text: str, index: dict[str, str]) -> list[str]: | |
| """Every term whose surface appears in `text`, word-boundary matched.""" | |
| haystack = normalize(text) | |
| if not haystack: | |
| return [] | |
| found: list[str] = [] | |
| for surface, term_id in index.items(): | |
| if term_id not in found and _word_match(surface, haystack): | |
| found.append(term_id) | |
| return found | |
| def link_all( | |
| glossary: list[dict], | |
| rules: list[dict], | |
| formulas: list[dict], | |
| clustered: ClusterResult, | |
| domain: dict | None = None, | |
| ) -> dict[str, int]: | |
| """Mutates the entries in place. Returns dangle counts for reporting. | |
| The counts are returned rather than logged-and-forgotten because a link | |
| stage that silently resolves nothing looks identical to one that works. | |
| """ | |
| index = _index_terms(glossary, clustered) | |
| # formulas: variables[].symbol -> term_id | |
| unresolved_vars = 0 | |
| for formula in formulas: | |
| for variable in formula.get("variables") or []: | |
| term_id = index.get(normalize(variable.get("symbol") or "")) | |
| variable["term_id"] = term_id | |
| if term_id is None: | |
| unresolved_vars += 1 | |
| # glossary: term -> the formula that DEFINES it (not one it appears in) | |
| for entry in glossary: | |
| entry["defining_formula_id"] = _defining_formula(entry, formulas) | |
| unresolved_defining = sum(1 for e in glossary if not e.get("defining_formula_id")) | |
| # rules: -> terms named in the rule text, and formulas from the same chunk | |
| formulas_by_chunk: dict[str, list[str]] = {} | |
| for formula in formulas: | |
| chunk_id = (formula.get("provenance") or {}).get("chunk_id") | |
| if chunk_id: | |
| formulas_by_chunk.setdefault(chunk_id, []).append(formula["formula_id"]) | |
| for rule in rules: | |
| text = " ".join( | |
| str(rule.get(f) or "") for f in ("statement", "condition", "consequence") | |
| ) | |
| rule["term_ids"] = _resolve(text, index) | |
| chunk_id = (rule.get("provenance") or {}).get("chunk_id") | |
| rule["formula_ids"] = list(formulas_by_chunk.get(chunk_id or "", [])) | |
| # domain context: key_parameters[].surface -> term_id | |
| # | |
| # Exact-surface lookup, not `_resolve`. A key parameter IS a term name, so | |
| # scanning it for every other term's surface would resolve "Physical | |
| # Availability (PA)" to whichever term matched first. The rules branch wants | |
| # "every term mentioned in this sentence"; this one wants "the term this | |
| # name IS", and they are different questions. | |
| unresolved_params = 0 | |
| for parameter in (domain or {}).get("key_parameters") or []: | |
| term_id = _match_parameter(parameter.get("surface") or "", index) | |
| parameter["term_id"] = term_id | |
| if term_id is None: | |
| unresolved_params += 1 | |
| return { | |
| "terms_indexed": len(index), | |
| "variables_unresolved": unresolved_vars, | |
| "glossary_without_defining_formula": unresolved_defining, | |
| "rules_without_term_link": sum(1 for r in rules if not r["term_ids"]), | |
| "key_parameters_unresolved": unresolved_params, | |
| } | |
| def _match_parameter(surface: str, index: dict[str, str]) -> str | None: | |
| """Resolve one key-parameter surface to a term, most specific form first. | |
| The document names its parameters as `Full Name (ABBR)` — "Physical | |
| Availability (PA)" — while the glossary is keyed on the abbreviation with | |
| the expansion in `full_name`. A single exact lookup therefore resolved | |
| NOTHING on the reference document: 11 terms indexed, all 5 parameters | |
| dangling, and five false hallucination signals sorted to the top of the | |
| review queue, above every real term. Measured 2026-09-02. | |
| Three candidate keys, tried longest-first so the most specific spelling | |
| wins: | |
| "Physical Availability (PA)" -> whole | |
| -> "Physical Availability" (expansion) | |
| -> "PA" (abbreviation) | |
| Still **exact lookups**, never a scan over every indexed surface: this asks | |
| "which term IS this name", and scanning would answer "which terms are | |
| mentioned in it" — a different question that resolves "Physical | |
| Availability (PA)" to whichever term happened to match first. | |
| The abbreviation fallback is not redundant with the expansion one. On the | |
| reference standard the document writes "Physical Availability (PA)" while | |
| the glossary's `full_name` reads "Physical **of** Availability", so the | |
| expansion misses and only the abbreviation connects them — the same wording | |
| disagreement the pipeline is required to surface rather than normalise. | |
| """ | |
| whole = normalize(surface) | |
| if not whole: | |
| return None | |
| candidates = [whole] | |
| if "(" in surface and ")" in surface: | |
| head, _, tail = surface.partition("(") | |
| expansion = normalize(head) | |
| abbrev = normalize(tail.partition(")")[0]) | |
| candidates += [c for c in (expansion, abbrev) if c] | |
| for candidate in candidates: | |
| term_id = index.get(candidate) | |
| if term_id is not None: | |
| return term_id | |
| return None | |
| def _defining_formula(entry: dict, formulas: list[dict]) -> str | None: | |
| """The formula this term is defined BY, or None. | |
| Two signals, both deterministic: the formula's `name` names the term, or the | |
| term sits on the left of the `=`. Appearing on the right means the term is | |
| an *input* to that formula, which is the edge deliberately left derivable | |
| rather than stored. | |
| """ | |
| surfaces = [ | |
| normalize(entry.get("term") or ""), | |
| normalize(entry.get("full_name") or ""), | |
| ] | |
| surfaces = [s for s in surfaces if s] | |
| if not surfaces: | |
| return None | |
| for formula in formulas: | |
| name = normalize(formula.get("name") or "") | |
| if name and any(_word_match(s, name) for s in surfaces): | |
| return formula["formula_id"] | |
| latex = formula.get("formula_latex") or "" | |
| if "=" in latex: | |
| lhs = normalize(latex.split("=", 1)[0]) | |
| if lhs and any(_word_match(s, lhs) for s in surfaces): | |
| return formula["formula_id"] | |
| return None | |