"""Render a `DomainContext` as the text an agent actually reads (K2). Two tiers, because the preload-vs-lookup question is settled by size and hit-rate rather than by kind (KNOWLEDGE_CONSUMPTION_PLAN.md §2): - `render_card` is **Tier 1**, preloaded beside the catalog. Small, almost always relevant: what the domain is, its measures with their formulas and governing rules, and the full vocabulary as bare surfaces. - `render_lookup` is **Tier 2**, fetched for the two or three terms a question actually touches. **The vocabulary list is the point of the card.** A list of term surfaces costs a few hundred tokens and is what stops a planner force-mapping a term it does not recognise onto a column that merely looks similar - the pr/13 failure where `pa` was aliased as "revenue". Knowing the vocabulary EXISTS prevents that; knowing every definition is not required for it. Ceilings mirror `agents/planner/inputs.py` (F-12): safety nets set high, not tight caps, with a truncation line so the reader knows it saw a subset and a log so we learn when to revisit them. """ from __future__ import annotations from src.knowledge_domain.models import DomainContext, Measure from src.middlewares.logging import get_logger logger = get_logger("knowledge_render") # Sized from the measured benchmarks behind this design: a ~4 KB semantic # context document moved text-to-SQL accuracy from ~46-51% to ~68-69%. 12k chars # is roughly 3k tokens - generous next to the catalog's 250k, because this # content is denser per token and a domain that needs more than this is telling # us the corpus has outgrown a single card rather than that the cap is wrong. MAX_CARD_CHARS = 12_000 # Beyond this the card stops being a card. Measures are ordered by established # usage, so a cut takes the least-used first. MAX_MEASURES = 40 # The vocabulary list is cheap per entry; this only guards a runaway corpus. MAX_SURFACES = 400 def _measure_lines(measure: Measure) -> list[str]: """One measure, at the density a planner needs to decide with.""" head = measure.name aliases = [f for f in measure.surface_forms if f != measure.name] if aliases: head += f" (also: {', '.join(aliases)})" if measure.unit: head += f" [{measure.unit}]" lines = [f"- {head}"] if measure.definition: lines.append(f" {measure.definition.strip()}") if measure.formula_latex: lines.append(f" formula: {measure.formula_latex.strip()}") if measure.inputs: lines.append(f" inputs: {', '.join(measure.inputs)}") if measure.catalog_binding: # The whole point of binding: the planner can go straight to the column # instead of guessing which one the term means. lines.append(f" column: {measure.catalog_binding}") if measure.governed_by: lines.append( f" governed by {len(measure.governed_by)} rule(s) — look them up " f"before computing this" ) if measure.contested: # Never resolved silently. A planner must know the documents disagree. lines.append( f" ⚠ DISPUTED: {len(measure.definitions)} documents define this " f"differently — do not assume one reading" ) return lines def render_card(context: DomainContext) -> str: """Tier 1: the block that goes into the planner prompt beside the catalog. Returns `""` for an empty context rather than a header with nothing under it — a card that says nothing still costs tokens and still implies an authority it does not have. """ if context.is_empty: return "" out: list[str] = ["## Domain knowledge"] identity = context.identity if identity.domain_name: out.append(f"Domain: {identity.domain_name}") if identity.subdomains: out.append(f"Covers: {', '.join(identity.subdomains)}") if identity.boundary: # The field that makes refusal first-class - see the v4 plan §3.1. A # question landing in one of these has a DOMAIN gap, which is a # different and more honest answer than a data gap. out.append(f"No approved knowledge about: {identity.boundary}") out.append( "A question about those is a DOMAIN gap — say so rather than " "answering from a column that happens to exist." ) if identity.boundary_note: out.append(identity.boundary_note) conventions = context.conventions if conventions.time_grain: out.append(f"Time grain: {conventions.time_grain}") if conventions.units: out.append(f"Units: {', '.join(u.symbol for u in conventions.units)}") measures = context.measures[:MAX_MEASURES] if measures: out.append("") out.append("### Measures") for measure in measures: out.extend(_measure_lines(measure)) dropped = len(context.measures) - len(measures) if dropped > 0: out.append(f"- … {dropped} less-used measure(s) not shown; look them up by name") # The vocabulary index: every surface, no definitions. Cheap, and the thing # that stops a term being force-mapped just because it was unrecognised. surfaces: list[str] = [] for measure in context.measures: for form in measure.surface_forms: if form not in surfaces: surfaces.append(form) if surfaces: shown = surfaces[:MAX_SURFACES] out.append("") out.append("### Known terms") out.append(", ".join(shown)) if len(surfaces) > len(shown): out.append(f"… and {len(surfaces) - len(shown)} more") out.append( "A term above that you cannot map to a column is a DATA GAP, not an " "invitation to substitute a similar column." ) if context.policies: out.append("") out.append( f"### Rules\n{len(context.policies)} domain-wide rule(s) apply. Look " f"them up before relying on a calculation." ) coverage = context.authority.coverage if coverage.n_documents: # Says how much authority this card actually has. A sparse context # should be weighed differently, and hiding that would be dishonest. note = ( f"Built from {coverage.n_documents} reviewed document(s), " f"{coverage.n_measures} measure(s)." ) if not coverage.declared: note += " Scope and boundary not yet declared by an expert." out.append("") out.append(note) card = "\n".join(out) if len(card) > MAX_CARD_CHARS: card = card[:MAX_CARD_CHARS].rsplit("\n", 1)[0] card += "\n… domain card truncated; look up specific terms as needed." logger.warning( "domain_card_truncated", extra={"scope_id": context.scope_id, "chars": len(card), "measures": len(context.measures)}, ) return card def render_lookup(measures: list[Measure], rules: list[dict] | None = None) -> str: """Tier 2: everything known about the few terms a question touched. No ceiling: the caller chose these, and a lookup that silently drops what was asked for is worse than a long one. """ if not measures and not rules: return "No domain knowledge recorded for those terms." out: list[str] = [] for measure in measures: out.append(f"### {measure.name}") if measure.surface_forms: out.append(f"Also written: {', '.join(measure.surface_forms)}") for definition in measure.definitions or []: out.append(f"- {definition.text.strip()} [{definition.doc_id}]") if not measure.definitions and measure.definition: out.append(f"- {measure.definition.strip()}") if measure.formula_latex: out.append(f"Formula: {measure.formula_latex.strip()}") if measure.unit: out.append(f"Unit: {measure.unit}") if measure.catalog_binding: out.append(f"Column: {measure.catalog_binding}") if measure.contested: out.append( "⚠ The documents DISAGREE about this term. Both readings are " "above; say which you used." ) out.append("") for rule in rules or []: statement = rule.get("statement") or "" condition, consequence = rule.get("condition"), rule.get("consequence") out.append("### Rule") if statement: out.append(statement.strip()) if condition and consequence: out.append(f"IF {condition.strip()} THEN {consequence.strip()}") out.append("") return "\n".join(out).strip()