"""Domain Context v4 - the scope-level semantic model. Design and rationale: `docs/knowledge/KNOWLEDGE_DOMAIN_CONTEXT_V4.md`. **One per company (`scope_id`), composed from that company's approved entries.** This is the object the name always promised: not a description of a file, but a model of the business an agent can reason about. v3's per-document object keeps its fields and becomes a `DocumentBrief`, demoted to a source record. The split that matters, and the reason this is cheap to build: - **Derived** - `measures`, `surface_forms`, `policies`, `units`, `coverage` all fall out of entries the pipeline already extracts. Deterministic, checkable, and no prompt is involved. - **Declared** - `identity`, `boundary`, `conventions`, `entities`, `dimensions` are the TBox: the model proposes and an expert approves them through the review loop (decided 2026-09-08). Absent until someone rules on them, which is why every one of those fields is Optional and the object is useful without them. `catalog_binding` is deliberately **not stored** - it is resolved at read time against whichever catalog the caller has, the same stopgap shape as `fk_inference.py`, and self-disabling when a binding cannot be made. """ from __future__ import annotations from datetime import datetime from pydantic import BaseModel, Field class Definition(BaseModel): """One document's account of a measure, with the document that said it. A list of these, rather than a single winning string, because when two documents disagree the disagreement IS the finding. Picking a winner is an expert judgement and the pipeline exists to put that in front of a human. """ text: str doc_id: str term_id: str approved: bool = False class Measure(BaseModel): """One thing the business counts, assembled from three entry kinds. A measure is the join that makes the pipeline's output usable: the glossary says what it *is*, the formula says how it is *computed*, and the rules say when that computation is *valid*. Any one alone is documentation; together they are something a planner can act on. """ # Stable across documents: the canonical surface, normalised by the same # rule that clusters mentions inside one document. `term_id` is # document-scoped by design, which is right for review and wrong here - two # documents defining PA would otherwise be two measures. key: str # Every document-scoped id that contributed, for provenance back to review. term_ids: list[str] = Field(default_factory=list) name: str # Every way the documents write it, including the ones that disagree with # each other - the BUMA standard heads a section "Physical *of* Availability # (PA)" while its legend says "Physical Availability". Surfacing the # disagreement is a locked product decision, so both forms travel together. surface_forms: list[str] = Field(default_factory=list) definition: str | None = None # Populated only when documents DISAGREE. One entry means one account and # `definition` carries it; more than one means the measure is contested and # a consumer must not pick silently. definitions: list[Definition] = Field(default_factory=list) contested: bool = False source_docs: list[str] = Field(default_factory=list) unit: str | None = None grain: str | None = None formula_id: str | None = None formula_latex: str | None = None # What the formula consumes, by symbol. Resolved to `term_id`s where the # linker managed it; a bare symbol is expected and not a failure. inputs: list[str] = Field(default_factory=list) # Rules that constrain this measure - "use joint-survey production, else # truck count". This is the field that turns a correct calculation on the # wrong input into a correct answer. governed_by: list[str] = Field(default_factory=list) # Resolved at read time when a catalog is supplied. None means the measure # is described but not bound to any column the user actually has. catalog_binding: str | None = None mention_count: int = 0 # False when the entry is being read ahead of expert approval (dev only). approved: bool = False class DomainEntity(BaseModel): """A thing the business measures - a unit, a shift, a site. Declared.""" name: str grain: str | None = None synonyms: list[str] = Field(default_factory=list) catalog_binding: str | None = None class Dimension(BaseModel): """How measures are sliced. Declared.""" name: str synonyms: list[str] = Field(default_factory=list) catalog_binding: str | None = None class Unit(BaseModel): symbol: str meaning: str | None = None class Conventions(BaseModel): """How the operator counts time and quantity. Declared, except `units`.""" time_grain: str | None = None calendar: str | None = None units: list[Unit] = Field(default_factory=list) class Identity(BaseModel): """What this domain is, and what it is not.""" domain_name: str | None = None subdomains: list[str] = Field(default_factory=list) # What the domain does NOT cover. The field that makes refusal first-class: # the planner's honest-data-gap path fires today when the CATALOG cannot # answer, and should also fire when the DOMAIN cannot. "This standard # governs equipment availability, not cost" is a better answer than a forced # mapping onto a cost column that happens to exist. boundary: str | None = None # The expert's elaboration, when they add one. Free text, so it is theirs # to write and never derived. boundary_note: str | None = None class SourceRef(BaseModel): """One document this context was built from.""" doc_id: str title: str | None = None n_pages: int = 0 approved_at: datetime | None = None class Coverage(BaseModel): """How much of this domain is actually populated and reviewed. The honesty field. A context assembled from one 9-page standard must not present itself with the authority of a reviewed library, and a consumer should be able to weigh it accordingly. """ n_documents: int = 0 n_measures: int = 0 n_policies: int = 0 n_entries_total: int = 0 n_entries_approved: int = 0 # None when nothing has been reviewed - deliberately not 0.0, because "no # reviews yet" and "everything was rejected" are different situations. approved_ratio: float | None = None declared: bool = False class Authority(BaseModel): sources: list[SourceRef] = Field(default_factory=list) as_of: datetime | None = None coverage: Coverage = Field(default_factory=Coverage) class DomainContext(BaseModel): """The scope-level semantic model. One per company.""" scope_id: str identity: Identity = Field(default_factory=Identity) measures: list[Measure] = Field(default_factory=list) entities: list[DomainEntity] = Field(default_factory=list) dimensions: list[Dimension] = Field(default_factory=list) conventions: Conventions = Field(default_factory=Conventions) # Domain-wide rules: approved rules governing no specific term. A rule tied # to a measure travels on that measure instead, so nothing is listed twice. policies: list[str] = Field(default_factory=list) authority: Authority = Field(default_factory=Authority) @property def is_empty(self) -> bool: """No measures and nothing declared - there is nothing to give a planner. Worth checking explicitly rather than letting an empty card into a prompt: a domain context that says nothing still costs tokens and still implies authority it does not have. """ return not self.measures and not self.identity.domain_name