ishaq101's picture
/fix parsing and term extract (#21)
f07443e
Raw History Blame Contribute Delete
6.5 kB
"""Readable concept keys — `TYPE_SUBJECT`, the label a person or an agent searches by.
Asked for in the 2026-09-08 discussion (m1). The shape is the one agreed there:
the type first, then what the entry is *about*, in capitals — `GLOSSARY_PA`,
never the entry's full text. Two properties it exists to provide:
- **One concept, one key.** `PA` drawn from five documents still reads
`GLOSSARY_PA`, so an agent that needs a term's definition writes the key and
looks it up directly — no scan, no embedding, no model call.
- **Readable at a glance.** Capitals were a deliberate call (lowercase reads
worse in a list), and lookups are capitals too.
**A key is a label, never an identity — this is m2.** `entity_id` in `ids.py`
stays the thing an expert's approval hangs on, because it is a hash of content
and survives a re-run untouched. A key is built from a *surface*, so it moves
the moment clustering canonicalises that surface differently. Both travel
together and neither substitutes for the other: the hash identifies, the key
reads. Collapsing the two is exactly how approvals get orphaned, which is the
failure `ids.py` was written to prevent.
Prefixes use the `kind` column's own vocabulary (English, singular) so the
system holds one spelling of each branch rather than two to keep in sync
(decided 2026-09-14).
"""
from __future__ import annotations
import hashlib
import re
import unicodedata
from collections.abc import Iterable, Mapping
from typing import Any
# Keyed by the `kind` column's values, so a new kind that forgets to register
# here returns None rather than silently inventing a prefix.
_PREFIX: dict[str, str] = {
"glossary": "GLOSSARY",
"rule": "RULE",
"formula": "FORMULA",
"document": "DOCUMENT",
"domain": "DOMAIN",
}
# Past this a subject stops being a label and becomes the content it was meant
# to stand in for. A rule's `statement` can be a whole paragraph.
MAX_SUBJECT = 40
# Short, and deliberately NOT the full `entity_id`: this digest separates two
# rules that govern the same term. It does not identify the row — `entity_id`
# does, and shortening a hash for readability is only safe because of that.
DIGEST_LEN = 6
def slugify(surface: str) -> str:
"""`Physical of Availability (PA)` -> `PHYSICAL_OF_AVAILABILITY_PA`.
Accents are folded rather than dropped so an Indonesian or a borrowed term
keeps its letters. Everything that is not a letter or a digit becomes a
single underscore, because a key that carries brackets or slashes cannot be
typed from memory — which is the whole point of having one.
"""
folded = unicodedata.normalize("NFKD", surface or "")
folded = "".join(c for c in folded if not unicodedata.combining(c))
slug = re.sub(r"[^A-Za-z0-9]+", "_", folded).upper()
slug = re.sub(r"_+", "_", slug).strip("_")
if len(slug) <= MAX_SUBJECT:
return slug
# Cut on a word boundary so the tail is not half a word — but a single long
# token has no boundary to cut on, and an empty subject is worse than a
# blunt one.
cut = slug[:MAX_SUBJECT]
head, _, _tail = cut.rpartition("_")
return (head or cut).strip("_")
def _digest(value: str) -> str:
return hashlib.sha256(value.encode("utf-8")).hexdigest()[:DIGEST_LEN].upper()
def term_surfaces(glossary_payloads: Iterable[Mapping[str, Any]] | None) -> dict[str, str]:
"""`term_id` -> the term's surface, for keying the rules that govern it.
Built by the caller from the same run's glossary payloads, because a rule
payload carries `term_ids` and no surfaces — the words live on the glossary
entry, not on the rule.
"""
out: dict[str, str] = {}
for payload in glossary_payloads or []:
term_id = payload.get("term_id")
term = payload.get("term")
if term_id and isinstance(term, str) and term.strip():
out[str(term_id)] = term.strip()
return out
def _rule_subject(
payload: Mapping[str, Any], entity_id: str, surfaces: Mapping[str, str]
) -> str:
"""A rule has no canonical surface of its own — it is a sentence.
So it borrows the term it governs and adds a short digest, because one term
usually governs several rules: `RULE_PA_C07BE4`. Asking the model to name
the rule was the obvious alternative and is the one `ids.py` already rules
out — a model cannot produce a stable identifier, and a model-written
`rule_id` was a real defect here, returning a different slug for the same
rule on consecutive runs at temperature 0. A rule that governs no term
keeps the digest alone; that it is not searchable by word was accepted when
this was agreed.
"""
for term_id in payload.get("term_ids") or []:
surface = slugify(surfaces.get(str(term_id), ""))
if surface:
return f"{surface}_{_digest(entity_id)}"
return _digest(entity_id)
def _subject(
kind: str,
payload: Mapping[str, Any],
entity_id: str,
doc_id: str,
surfaces: Mapping[str, str],
) -> str:
if kind == "glossary":
return slugify(payload.get("term") or "")
if kind == "formula":
# A formula may legitimately have no name; `entity_id` already falls
# back name -> latex -> chunk_id, so the digest inherits that work.
return slugify(payload.get("name") or "") or _digest(entity_id)
if kind == "document":
# The document's own id reads better than its title. A title is a
# verbatim transcription and routinely carries the PDF's broken
# spacing — correct to store, useless to type.
return slugify(doc_id) or _digest(entity_id)
if kind == "domain":
return slugify(payload.get("domain_name") or "")
if kind == "rule":
return _rule_subject(payload, entity_id, surfaces)
return ""
def concept_key(
kind: str,
payload: Mapping[str, Any],
*,
entity_id: str,
doc_id: str = "",
surfaces: Mapping[str, str] | None = None,
) -> str | None:
"""The `TYPE_SUBJECT` key for one entry; None when the kind is unregistered.
Never raises: the caller writes this into a row alongside content it has
already validated, and a key that cannot be built is a degraded label, not
a reason to lose the entry.
"""
prefix = _PREFIX.get(kind)
if prefix is None:
return None
subject = _subject(kind, payload, entity_id, doc_id, surfaces or {})
return f"{prefix}_{subject}" if subject else prefix