Download src/knowledge_extraction/keys.py from DataEyond/Agentic-Service-Data-Eyond-Catalog: direct link, hf CLI and curl.
- Browser
- Download file 6.5 kB
-
https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_extraction/keys.py
- Command line
-
hf download hf://spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/src/knowledge_extraction/keys.py
-
curl -L -o keys.py https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_extraction/keys.py
6.5 kB
| """Readable concept keys — `TYPE_SUBJECT`, the label a person or an agent searches by. | |
| Asked for in the 2026-09-08 discussion (m1). The shape is the one agreed there: | |
| the type first, then what the entry is *about*, in capitals — `GLOSSARY_PA`, | |
| never the entry's full text. Two properties it exists to provide: | |
| - **One concept, one key.** `PA` drawn from five documents still reads | |
| `GLOSSARY_PA`, so an agent that needs a term's definition writes the key and | |
| looks it up directly — no scan, no embedding, no model call. | |
| - **Readable at a glance.** Capitals were a deliberate call (lowercase reads | |
| worse in a list), and lookups are capitals too. | |
| **A key is a label, never an identity — this is m2.** `entity_id` in `ids.py` | |
| stays the thing an expert's approval hangs on, because it is a hash of content | |
| and survives a re-run untouched. A key is built from a *surface*, so it moves | |
| the moment clustering canonicalises that surface differently. Both travel | |
| together and neither substitutes for the other: the hash identifies, the key | |
| reads. Collapsing the two is exactly how approvals get orphaned, which is the | |
| failure `ids.py` was written to prevent. | |
| Prefixes use the `kind` column's own vocabulary (English, singular) so the | |
| system holds one spelling of each branch rather than two to keep in sync | |
| (decided 2026-09-14). | |
| """ | |
| from __future__ import annotations | |
| import hashlib | |
| import re | |
| import unicodedata | |
| from collections.abc import Iterable, Mapping | |
| from typing import Any | |
| # Keyed by the `kind` column's values, so a new kind that forgets to register | |
| # here returns None rather than silently inventing a prefix. | |
| _PREFIX: dict[str, str] = { | |
| "glossary": "GLOSSARY", | |
| "rule": "RULE", | |
| "formula": "FORMULA", | |
| "document": "DOCUMENT", | |
| "domain": "DOMAIN", | |
| } | |
| # Past this a subject stops being a label and becomes the content it was meant | |
| # to stand in for. A rule's `statement` can be a whole paragraph. | |
| MAX_SUBJECT = 40 | |
| # Short, and deliberately NOT the full `entity_id`: this digest separates two | |
| # rules that govern the same term. It does not identify the row — `entity_id` | |
| # does, and shortening a hash for readability is only safe because of that. | |
| DIGEST_LEN = 6 | |
| def slugify(surface: str) -> str: | |
| """`Physical of Availability (PA)` -> `PHYSICAL_OF_AVAILABILITY_PA`. | |
| Accents are folded rather than dropped so an Indonesian or a borrowed term | |
| keeps its letters. Everything that is not a letter or a digit becomes a | |
| single underscore, because a key that carries brackets or slashes cannot be | |
| typed from memory — which is the whole point of having one. | |
| """ | |
| folded = unicodedata.normalize("NFKD", surface or "") | |
| folded = "".join(c for c in folded if not unicodedata.combining(c)) | |
| slug = re.sub(r"[^A-Za-z0-9]+", "_", folded).upper() | |
| slug = re.sub(r"_+", "_", slug).strip("_") | |
| if len(slug) <= MAX_SUBJECT: | |
| return slug | |
| # Cut on a word boundary so the tail is not half a word — but a single long | |
| # token has no boundary to cut on, and an empty subject is worse than a | |
| # blunt one. | |
| cut = slug[:MAX_SUBJECT] | |
| head, _, _tail = cut.rpartition("_") | |
| return (head or cut).strip("_") | |
| def _digest(value: str) -> str: | |
| return hashlib.sha256(value.encode("utf-8")).hexdigest()[:DIGEST_LEN].upper() | |
| def term_surfaces(glossary_payloads: Iterable[Mapping[str, Any]] | None) -> dict[str, str]: | |
| """`term_id` -> the term's surface, for keying the rules that govern it. | |
| Built by the caller from the same run's glossary payloads, because a rule | |
| payload carries `term_ids` and no surfaces — the words live on the glossary | |
| entry, not on the rule. | |
| """ | |
| out: dict[str, str] = {} | |
| for payload in glossary_payloads or []: | |
| term_id = payload.get("term_id") | |
| term = payload.get("term") | |
| if term_id and isinstance(term, str) and term.strip(): | |
| out[str(term_id)] = term.strip() | |
| return out | |
| def _rule_subject( | |
| payload: Mapping[str, Any], entity_id: str, surfaces: Mapping[str, str] | |
| ) -> str: | |
| """A rule has no canonical surface of its own — it is a sentence. | |
| So it borrows the term it governs and adds a short digest, because one term | |
| usually governs several rules: `RULE_PA_C07BE4`. Asking the model to name | |
| the rule was the obvious alternative and is the one `ids.py` already rules | |
| out — a model cannot produce a stable identifier, and a model-written | |
| `rule_id` was a real defect here, returning a different slug for the same | |
| rule on consecutive runs at temperature 0. A rule that governs no term | |
| keeps the digest alone; that it is not searchable by word was accepted when | |
| this was agreed. | |
| """ | |
| for term_id in payload.get("term_ids") or []: | |
| surface = slugify(surfaces.get(str(term_id), "")) | |
| if surface: | |
| return f"{surface}_{_digest(entity_id)}" | |
| return _digest(entity_id) | |
| def _subject( | |
| kind: str, | |
| payload: Mapping[str, Any], | |
| entity_id: str, | |
| doc_id: str, | |
| surfaces: Mapping[str, str], | |
| ) -> str: | |
| if kind == "glossary": | |
| return slugify(payload.get("term") or "") | |
| if kind == "formula": | |
| # A formula may legitimately have no name; `entity_id` already falls | |
| # back name -> latex -> chunk_id, so the digest inherits that work. | |
| return slugify(payload.get("name") or "") or _digest(entity_id) | |
| if kind == "document": | |
| # The document's own id reads better than its title. A title is a | |
| # verbatim transcription and routinely carries the PDF's broken | |
| # spacing — correct to store, useless to type. | |
| return slugify(doc_id) or _digest(entity_id) | |
| if kind == "domain": | |
| return slugify(payload.get("domain_name") or "") | |
| if kind == "rule": | |
| return _rule_subject(payload, entity_id, surfaces) | |
| return "" | |
| def concept_key( | |
| kind: str, | |
| payload: Mapping[str, Any], | |
| *, | |
| entity_id: str, | |
| doc_id: str = "", | |
| surfaces: Mapping[str, str] | None = None, | |
| ) -> str | None: | |
| """The `TYPE_SUBJECT` key for one entry; None when the kind is unregistered. | |
| Never raises: the caller writes this into a row alongside content it has | |
| already validated, and a key that cannot be built is a degraded label, not | |
| a reason to lose the entry. | |
| """ | |
| prefix = _PREFIX.get(kind) | |
| if prefix is None: | |
| return None | |
| subject = _subject(kind, payload, entity_id, doc_id, surfaces or {}) | |
| return f"{prefix}_{subject}" if subject else prefix | |