Download src/knowledge_domain/models.py from DataEyond/Agentic-Service-Data-Eyond-Catalog: direct link, hf CLI and curl.
- Browser
- Download file 7.84 kB
-
https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_domain/models.py
- Command line
-
hf download hf://spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/src/knowledge_domain/models.py
-
curl -L -o models.py https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_domain/models.py
7.84 kB
| """Domain Context v4 - the scope-level semantic model. | |
| Design and rationale: `docs/knowledge/KNOWLEDGE_DOMAIN_CONTEXT_V4.md`. | |
| **One per company (`scope_id`), composed from that company's approved entries.** | |
| This is the object the name always promised: not a description of a file, but a | |
| model of the business an agent can reason about. v3's per-document object keeps | |
| its fields and becomes a `DocumentBrief`, demoted to a source record. | |
| The split that matters, and the reason this is cheap to build: | |
| - **Derived** - `measures`, `surface_forms`, `policies`, `units`, `coverage` all | |
| fall out of entries the pipeline already extracts. Deterministic, checkable, | |
| and no prompt is involved. | |
| - **Declared** - `identity`, `boundary`, `conventions`, `entities`, `dimensions` | |
| are the TBox: the model proposes and an expert approves them through the | |
| review loop (decided 2026-09-08). Absent until someone rules on them, which is | |
| why every one of those fields is Optional and the object is useful without | |
| them. | |
| `catalog_binding` is deliberately **not stored** - it is resolved at read time | |
| against whichever catalog the caller has, the same stopgap shape as | |
| `fk_inference.py`, and self-disabling when a binding cannot be made. | |
| """ | |
| from __future__ import annotations | |
| from datetime import datetime | |
| from pydantic import BaseModel, Field | |
| class Definition(BaseModel): | |
| """One document's account of a measure, with the document that said it. | |
| A list of these, rather than a single winning string, because when two | |
| documents disagree the disagreement IS the finding. Picking a winner is an | |
| expert judgement and the pipeline exists to put that in front of a human. | |
| """ | |
| text: str | |
| doc_id: str | |
| term_id: str | |
| approved: bool = False | |
| class Measure(BaseModel): | |
| """One thing the business counts, assembled from three entry kinds. | |
| A measure is the join that makes the pipeline's output usable: the glossary | |
| says what it *is*, the formula says how it is *computed*, and the rules say | |
| when that computation is *valid*. Any one alone is documentation; together | |
| they are something a planner can act on. | |
| """ | |
| # Stable across documents: the canonical surface, normalised by the same | |
| # rule that clusters mentions inside one document. `term_id` is | |
| # document-scoped by design, which is right for review and wrong here - two | |
| # documents defining PA would otherwise be two measures. | |
| key: str | |
| # Every document-scoped id that contributed, for provenance back to review. | |
| term_ids: list[str] = Field(default_factory=list) | |
| name: str | |
| # Every way the documents write it, including the ones that disagree with | |
| # each other - the BUMA standard heads a section "Physical *of* Availability | |
| # (PA)" while its legend says "Physical Availability". Surfacing the | |
| # disagreement is a locked product decision, so both forms travel together. | |
| surface_forms: list[str] = Field(default_factory=list) | |
| definition: str | None = None | |
| # Populated only when documents DISAGREE. One entry means one account and | |
| # `definition` carries it; more than one means the measure is contested and | |
| # a consumer must not pick silently. | |
| definitions: list[Definition] = Field(default_factory=list) | |
| contested: bool = False | |
| source_docs: list[str] = Field(default_factory=list) | |
| unit: str | None = None | |
| grain: str | None = None | |
| formula_id: str | None = None | |
| formula_latex: str | None = None | |
| # What the formula consumes, by symbol. Resolved to `term_id`s where the | |
| # linker managed it; a bare symbol is expected and not a failure. | |
| inputs: list[str] = Field(default_factory=list) | |
| # Rules that constrain this measure - "use joint-survey production, else | |
| # truck count". This is the field that turns a correct calculation on the | |
| # wrong input into a correct answer. | |
| governed_by: list[str] = Field(default_factory=list) | |
| # Resolved at read time when a catalog is supplied. None means the measure | |
| # is described but not bound to any column the user actually has. | |
| catalog_binding: str | None = None | |
| mention_count: int = 0 | |
| # False when the entry is being read ahead of expert approval (dev only). | |
| approved: bool = False | |
| class DomainEntity(BaseModel): | |
| """A thing the business measures - a unit, a shift, a site. Declared.""" | |
| name: str | |
| grain: str | None = None | |
| synonyms: list[str] = Field(default_factory=list) | |
| catalog_binding: str | None = None | |
| class Dimension(BaseModel): | |
| """How measures are sliced. Declared.""" | |
| name: str | |
| synonyms: list[str] = Field(default_factory=list) | |
| catalog_binding: str | None = None | |
| class Unit(BaseModel): | |
| symbol: str | |
| meaning: str | None = None | |
| class Conventions(BaseModel): | |
| """How the operator counts time and quantity. Declared, except `units`.""" | |
| time_grain: str | None = None | |
| calendar: str | None = None | |
| units: list[Unit] = Field(default_factory=list) | |
| class Identity(BaseModel): | |
| """What this domain is, and what it is not.""" | |
| domain_name: str | None = None | |
| subdomains: list[str] = Field(default_factory=list) | |
| # What the domain does NOT cover. The field that makes refusal first-class: | |
| # the planner's honest-data-gap path fires today when the CATALOG cannot | |
| # answer, and should also fire when the DOMAIN cannot. "This standard | |
| # governs equipment availability, not cost" is a better answer than a forced | |
| # mapping onto a cost column that happens to exist. | |
| boundary: str | None = None | |
| # The expert's elaboration, when they add one. Free text, so it is theirs | |
| # to write and never derived. | |
| boundary_note: str | None = None | |
| class SourceRef(BaseModel): | |
| """One document this context was built from.""" | |
| doc_id: str | |
| title: str | None = None | |
| n_pages: int = 0 | |
| approved_at: datetime | None = None | |
| class Coverage(BaseModel): | |
| """How much of this domain is actually populated and reviewed. | |
| The honesty field. A context assembled from one 9-page standard must not | |
| present itself with the authority of a reviewed library, and a consumer | |
| should be able to weigh it accordingly. | |
| """ | |
| n_documents: int = 0 | |
| n_measures: int = 0 | |
| n_policies: int = 0 | |
| n_entries_total: int = 0 | |
| n_entries_approved: int = 0 | |
| # None when nothing has been reviewed - deliberately not 0.0, because "no | |
| # reviews yet" and "everything was rejected" are different situations. | |
| approved_ratio: float | None = None | |
| declared: bool = False | |
| class Authority(BaseModel): | |
| sources: list[SourceRef] = Field(default_factory=list) | |
| as_of: datetime | None = None | |
| coverage: Coverage = Field(default_factory=Coverage) | |
| class DomainContext(BaseModel): | |
| """The scope-level semantic model. One per company.""" | |
| scope_id: str | |
| identity: Identity = Field(default_factory=Identity) | |
| measures: list[Measure] = Field(default_factory=list) | |
| entities: list[DomainEntity] = Field(default_factory=list) | |
| dimensions: list[Dimension] = Field(default_factory=list) | |
| conventions: Conventions = Field(default_factory=Conventions) | |
| # Domain-wide rules: approved rules governing no specific term. A rule tied | |
| # to a measure travels on that measure instead, so nothing is listed twice. | |
| policies: list[str] = Field(default_factory=list) | |
| authority: Authority = Field(default_factory=Authority) | |
| def is_empty(self) -> bool: | |
| """No measures and nothing declared - there is nothing to give a planner. | |
| Worth checking explicitly rather than letting an empty card into a | |
| prompt: a domain context that says nothing still costs tokens and still | |
| implies authority it does not have. | |
| """ | |
| return not self.measures and not self.identity.domain_name | |