ishaq101's picture sofhiaazzhr's picture
/feat knowledge management (#20)
b68816f
Raw History Blame Contribute Delete
8.73 kB
"""Render a `DomainContext` as the text an agent actually reads (K2).
Two tiers, because the preload-vs-lookup question is settled by size and
hit-rate rather than by kind (KNOWLEDGE_CONSUMPTION_PLAN.md §2):
- `render_card` is **Tier 1**, preloaded beside the catalog. Small, almost always
relevant: what the domain is, its measures with their formulas and governing
rules, and the full vocabulary as bare surfaces.
- `render_lookup` is **Tier 2**, fetched for the two or three terms a question
actually touches.
**The vocabulary list is the point of the card.** A list of term surfaces costs a
few hundred tokens and is what stops a planner force-mapping a term it does not
recognise onto a column that merely looks similar - the pr/13 failure where `pa`
was aliased as "revenue". Knowing the vocabulary EXISTS prevents that; knowing
every definition is not required for it.
Ceilings mirror `agents/planner/inputs.py` (F-12): safety nets set high, not
tight caps, with a truncation line so the reader knows it saw a subset and a log
so we learn when to revisit them.
"""
from __future__ import annotations
from src.knowledge_domain.models import DomainContext, Measure
from src.middlewares.logging import get_logger
logger = get_logger("knowledge_render")
# Sized from the measured benchmarks behind this design: a ~4 KB semantic
# context document moved text-to-SQL accuracy from ~46-51% to ~68-69%. 12k chars
# is roughly 3k tokens - generous next to the catalog's 250k, because this
# content is denser per token and a domain that needs more than this is telling
# us the corpus has outgrown a single card rather than that the cap is wrong.
MAX_CARD_CHARS = 12_000
# Beyond this the card stops being a card. Measures are ordered by established
# usage, so a cut takes the least-used first.
MAX_MEASURES = 40
# The vocabulary list is cheap per entry; this only guards a runaway corpus.
MAX_SURFACES = 400
def _measure_lines(measure: Measure) -> list[str]:
"""One measure, at the density a planner needs to decide with."""
head = measure.name
aliases = [f for f in measure.surface_forms if f != measure.name]
if aliases:
head += f" (also: {', '.join(aliases)})"
if measure.unit:
head += f" [{measure.unit}]"
lines = [f"- {head}"]
if measure.definition:
lines.append(f" {measure.definition.strip()}")
if measure.formula_latex:
lines.append(f" formula: {measure.formula_latex.strip()}")
if measure.inputs:
lines.append(f" inputs: {', '.join(measure.inputs)}")
if measure.catalog_binding:
# The whole point of binding: the planner can go straight to the column
# instead of guessing which one the term means.
lines.append(f" column: {measure.catalog_binding}")
if measure.governed_by:
lines.append(
f" governed by {len(measure.governed_by)} rule(s) — look them up "
f"before computing this"
)
if measure.contested:
# Never resolved silently. A planner must know the documents disagree.
lines.append(
f" ⚠ DISPUTED: {len(measure.definitions)} documents define this "
f"differently — do not assume one reading"
)
return lines
def render_card(context: DomainContext) -> str:
"""Tier 1: the block that goes into the planner prompt beside the catalog.
Returns `""` for an empty context rather than a header with nothing under
it — a card that says nothing still costs tokens and still implies an
authority it does not have.
"""
if context.is_empty:
return ""
out: list[str] = ["## Domain knowledge"]
identity = context.identity
if identity.domain_name:
out.append(f"Domain: {identity.domain_name}")
if identity.subdomains:
out.append(f"Covers: {', '.join(identity.subdomains)}")
if identity.boundary:
# The field that makes refusal first-class - see the v4 plan §3.1. A
# question landing in one of these has a DOMAIN gap, which is a
# different and more honest answer than a data gap.
out.append(f"No approved knowledge about: {identity.boundary}")
out.append(
"A question about those is a DOMAIN gap — say so rather than "
"answering from a column that happens to exist."
)
if identity.boundary_note:
out.append(identity.boundary_note)
conventions = context.conventions
if conventions.time_grain:
out.append(f"Time grain: {conventions.time_grain}")
if conventions.units:
out.append(f"Units: {', '.join(u.symbol for u in conventions.units)}")
measures = context.measures[:MAX_MEASURES]
if measures:
out.append("")
out.append("### Measures")
for measure in measures:
out.extend(_measure_lines(measure))
dropped = len(context.measures) - len(measures)
if dropped > 0:
out.append(f"- … {dropped} less-used measure(s) not shown; look them up by name")
# The vocabulary index: every surface, no definitions. Cheap, and the thing
# that stops a term being force-mapped just because it was unrecognised.
surfaces: list[str] = []
for measure in context.measures:
for form in measure.surface_forms:
if form not in surfaces:
surfaces.append(form)
if surfaces:
shown = surfaces[:MAX_SURFACES]
out.append("")
out.append("### Known terms")
out.append(", ".join(shown))
if len(surfaces) > len(shown):
out.append(f"… and {len(surfaces) - len(shown)} more")
out.append(
"A term above that you cannot map to a column is a DATA GAP, not an "
"invitation to substitute a similar column."
)
if context.policies:
out.append("")
out.append(
f"### Rules\n{len(context.policies)} domain-wide rule(s) apply. Look "
f"them up before relying on a calculation."
)
coverage = context.authority.coverage
if coverage.n_documents:
# Says how much authority this card actually has. A sparse context
# should be weighed differently, and hiding that would be dishonest.
note = (
f"Built from {coverage.n_documents} reviewed document(s), "
f"{coverage.n_measures} measure(s)."
)
if not coverage.declared:
note += " Scope and boundary not yet declared by an expert."
out.append("")
out.append(note)
card = "\n".join(out)
if len(card) > MAX_CARD_CHARS:
card = card[:MAX_CARD_CHARS].rsplit("\n", 1)[0]
card += "\n… domain card truncated; look up specific terms as needed."
logger.warning(
"domain_card_truncated",
extra={"scope_id": context.scope_id, "chars": len(card),
"measures": len(context.measures)},
)
return card
def render_lookup(measures: list[Measure], rules: list[dict] | None = None) -> str:
"""Tier 2: everything known about the few terms a question touched.
No ceiling: the caller chose these, and a lookup that silently drops what was
asked for is worse than a long one.
"""
if not measures and not rules:
return "No domain knowledge recorded for those terms."
out: list[str] = []
for measure in measures:
out.append(f"### {measure.name}")
if measure.surface_forms:
out.append(f"Also written: {', '.join(measure.surface_forms)}")
for definition in measure.definitions or []:
out.append(f"- {definition.text.strip()} [{definition.doc_id}]")
if not measure.definitions and measure.definition:
out.append(f"- {measure.definition.strip()}")
if measure.formula_latex:
out.append(f"Formula: {measure.formula_latex.strip()}")
if measure.unit:
out.append(f"Unit: {measure.unit}")
if measure.catalog_binding:
out.append(f"Column: {measure.catalog_binding}")
if measure.contested:
out.append(
"⚠ The documents DISAGREE about this term. Both readings are "
"above; say which you used."
)
out.append("")
for rule in rules or []:
statement = rule.get("statement") or ""
condition, consequence = rule.get("condition"), rule.get("consequence")
out.append("### Rule")
if statement:
out.append(statement.strip())
if condition and consequence:
out.append(f"IF {condition.strip()} THEN {consequence.strip()}")
out.append("")
return "\n".join(out).strip()