"""Readable concept keys — `TYPE_SUBJECT`, the label a person or an agent searches by. Asked for in the 2026-09-08 discussion (m1). The shape is the one agreed there: the type first, then what the entry is *about*, in capitals — `GLOSSARY_PA`, never the entry's full text. Two properties it exists to provide: - **One concept, one key.** `PA` drawn from five documents still reads `GLOSSARY_PA`, so an agent that needs a term's definition writes the key and looks it up directly — no scan, no embedding, no model call. - **Readable at a glance.** Capitals were a deliberate call (lowercase reads worse in a list), and lookups are capitals too. **A key is a label, never an identity — this is m2.** `entity_id` in `ids.py` stays the thing an expert's approval hangs on, because it is a hash of content and survives a re-run untouched. A key is built from a *surface*, so it moves the moment clustering canonicalises that surface differently. Both travel together and neither substitutes for the other: the hash identifies, the key reads. Collapsing the two is exactly how approvals get orphaned, which is the failure `ids.py` was written to prevent. Prefixes use the `kind` column's own vocabulary (English, singular) so the system holds one spelling of each branch rather than two to keep in sync (decided 2026-09-14). """ from __future__ import annotations import hashlib import re import unicodedata from collections.abc import Iterable, Mapping from typing import Any # Keyed by the `kind` column's values, so a new kind that forgets to register # here returns None rather than silently inventing a prefix. _PREFIX: dict[str, str] = { "glossary": "GLOSSARY", "rule": "RULE", "formula": "FORMULA", "document": "DOCUMENT", "domain": "DOMAIN", } # Past this a subject stops being a label and becomes the content it was meant # to stand in for. A rule's `statement` can be a whole paragraph. MAX_SUBJECT = 40 # Short, and deliberately NOT the full `entity_id`: this digest separates two # rules that govern the same term. It does not identify the row — `entity_id` # does, and shortening a hash for readability is only safe because of that. DIGEST_LEN = 6 def slugify(surface: str) -> str: """`Physical of Availability (PA)` -> `PHYSICAL_OF_AVAILABILITY_PA`. Accents are folded rather than dropped so an Indonesian or a borrowed term keeps its letters. Everything that is not a letter or a digit becomes a single underscore, because a key that carries brackets or slashes cannot be typed from memory — which is the whole point of having one. """ folded = unicodedata.normalize("NFKD", surface or "") folded = "".join(c for c in folded if not unicodedata.combining(c)) slug = re.sub(r"[^A-Za-z0-9]+", "_", folded).upper() slug = re.sub(r"_+", "_", slug).strip("_") if len(slug) <= MAX_SUBJECT: return slug # Cut on a word boundary so the tail is not half a word — but a single long # token has no boundary to cut on, and an empty subject is worse than a # blunt one. cut = slug[:MAX_SUBJECT] head, _, _tail = cut.rpartition("_") return (head or cut).strip("_") def _digest(value: str) -> str: return hashlib.sha256(value.encode("utf-8")).hexdigest()[:DIGEST_LEN].upper() def term_surfaces(glossary_payloads: Iterable[Mapping[str, Any]] | None) -> dict[str, str]: """`term_id` -> the term's surface, for keying the rules that govern it. Built by the caller from the same run's glossary payloads, because a rule payload carries `term_ids` and no surfaces — the words live on the glossary entry, not on the rule. """ out: dict[str, str] = {} for payload in glossary_payloads or []: term_id = payload.get("term_id") term = payload.get("term") if term_id and isinstance(term, str) and term.strip(): out[str(term_id)] = term.strip() return out def _rule_subject( payload: Mapping[str, Any], entity_id: str, surfaces: Mapping[str, str] ) -> str: """A rule has no canonical surface of its own — it is a sentence. So it borrows the term it governs and adds a short digest, because one term usually governs several rules: `RULE_PA_C07BE4`. Asking the model to name the rule was the obvious alternative and is the one `ids.py` already rules out — a model cannot produce a stable identifier, and a model-written `rule_id` was a real defect here, returning a different slug for the same rule on consecutive runs at temperature 0. A rule that governs no term keeps the digest alone; that it is not searchable by word was accepted when this was agreed. """ for term_id in payload.get("term_ids") or []: surface = slugify(surfaces.get(str(term_id), "")) if surface: return f"{surface}_{_digest(entity_id)}" return _digest(entity_id) def _subject( kind: str, payload: Mapping[str, Any], entity_id: str, doc_id: str, surfaces: Mapping[str, str], ) -> str: if kind == "glossary": return slugify(payload.get("term") or "") if kind == "formula": # A formula may legitimately have no name; `entity_id` already falls # back name -> latex -> chunk_id, so the digest inherits that work. return slugify(payload.get("name") or "") or _digest(entity_id) if kind == "document": # The document's own id reads better than its title. A title is a # verbatim transcription and routinely carries the PDF's broken # spacing — correct to store, useless to type. return slugify(doc_id) or _digest(entity_id) if kind == "domain": return slugify(payload.get("domain_name") or "") if kind == "rule": return _rule_subject(payload, entity_id, surfaces) return "" def concept_key( kind: str, payload: Mapping[str, Any], *, entity_id: str, doc_id: str = "", surfaces: Mapping[str, str] | None = None, ) -> str | None: """The `TYPE_SUBJECT` key for one entry; None when the kind is unregistered. Never raises: the caller writes this into a row alongside content it has already validated, and a key that cannot be built is a degraded label, not a reason to lose the entry. """ prefix = _PREFIX.get(kind) if prefix is None: return None subject = _subject(kind, payload, entity_id, doc_id, surfaces or {}) return f"{prefix}_{subject}" if subject else prefix