File size: 8,731 Bytes
b68816f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
"""Render a `DomainContext` as the text an agent actually reads (K2).

Two tiers, because the preload-vs-lookup question is settled by size and
hit-rate rather than by kind (KNOWLEDGE_CONSUMPTION_PLAN.md §2):

- `render_card` is **Tier 1**, preloaded beside the catalog. Small, almost always
  relevant: what the domain is, its measures with their formulas and governing
  rules, and the full vocabulary as bare surfaces.
- `render_lookup` is **Tier 2**, fetched for the two or three terms a question
  actually touches.

**The vocabulary list is the point of the card.** A list of term surfaces costs a
few hundred tokens and is what stops a planner force-mapping a term it does not
recognise onto a column that merely looks similar - the pr/13 failure where `pa`
was aliased as "revenue". Knowing the vocabulary EXISTS prevents that; knowing
every definition is not required for it.

Ceilings mirror `agents/planner/inputs.py` (F-12): safety nets set high, not
tight caps, with a truncation line so the reader knows it saw a subset and a log
so we learn when to revisit them.
"""

from __future__ import annotations

from src.knowledge_domain.models import DomainContext, Measure
from src.middlewares.logging import get_logger

logger = get_logger("knowledge_render")

# Sized from the measured benchmarks behind this design: a ~4 KB semantic
# context document moved text-to-SQL accuracy from ~46-51% to ~68-69%. 12k chars
# is roughly 3k tokens - generous next to the catalog's 250k, because this
# content is denser per token and a domain that needs more than this is telling
# us the corpus has outgrown a single card rather than that the cap is wrong.
MAX_CARD_CHARS = 12_000
# Beyond this the card stops being a card. Measures are ordered by established
# usage, so a cut takes the least-used first.
MAX_MEASURES = 40
# The vocabulary list is cheap per entry; this only guards a runaway corpus.
MAX_SURFACES = 400


def _measure_lines(measure: Measure) -> list[str]:
    """One measure, at the density a planner needs to decide with."""
    head = measure.name
    aliases = [f for f in measure.surface_forms if f != measure.name]
    if aliases:
        head += f"  (also: {', '.join(aliases)})"
    if measure.unit:
        head += f"  [{measure.unit}]"
    lines = [f"- {head}"]

    if measure.definition:
        lines.append(f"    {measure.definition.strip()}")
    if measure.formula_latex:
        lines.append(f"    formula: {measure.formula_latex.strip()}")
    if measure.inputs:
        lines.append(f"    inputs: {', '.join(measure.inputs)}")
    if measure.catalog_binding:
        # The whole point of binding: the planner can go straight to the column
        # instead of guessing which one the term means.
        lines.append(f"    column: {measure.catalog_binding}")
    if measure.governed_by:
        lines.append(
            f"    governed by {len(measure.governed_by)} rule(s) — look them up "
            f"before computing this"
        )
    if measure.contested:
        # Never resolved silently. A planner must know the documents disagree.
        lines.append(
            f"    ⚠ DISPUTED: {len(measure.definitions)} documents define this "
            f"differently — do not assume one reading"
        )
    return lines


def render_card(context: DomainContext) -> str:
    """Tier 1: the block that goes into the planner prompt beside the catalog.

    Returns `""` for an empty context rather than a header with nothing under
    it — a card that says nothing still costs tokens and still implies an
    authority it does not have.
    """
    if context.is_empty:
        return ""

    out: list[str] = ["## Domain knowledge"]

    identity = context.identity
    if identity.domain_name:
        out.append(f"Domain: {identity.domain_name}")
    if identity.subdomains:
        out.append(f"Covers: {', '.join(identity.subdomains)}")
    if identity.boundary:
        # The field that makes refusal first-class - see the v4 plan §3.1. A
        # question landing in one of these has a DOMAIN gap, which is a
        # different and more honest answer than a data gap.
        out.append(f"No approved knowledge about: {identity.boundary}")
        out.append(
            "A question about those is a DOMAIN gap — say so rather than "
            "answering from a column that happens to exist."
        )
    if identity.boundary_note:
        out.append(identity.boundary_note)

    conventions = context.conventions
    if conventions.time_grain:
        out.append(f"Time grain: {conventions.time_grain}")
    if conventions.units:
        out.append(f"Units: {', '.join(u.symbol for u in conventions.units)}")

    measures = context.measures[:MAX_MEASURES]
    if measures:
        out.append("")
        out.append("### Measures")
        for measure in measures:
            out.extend(_measure_lines(measure))
    dropped = len(context.measures) - len(measures)
    if dropped > 0:
        out.append(f"- … {dropped} less-used measure(s) not shown; look them up by name")

    # The vocabulary index: every surface, no definitions. Cheap, and the thing
    # that stops a term being force-mapped just because it was unrecognised.
    surfaces: list[str] = []
    for measure in context.measures:
        for form in measure.surface_forms:
            if form not in surfaces:
                surfaces.append(form)
    if surfaces:
        shown = surfaces[:MAX_SURFACES]
        out.append("")
        out.append("### Known terms")
        out.append(", ".join(shown))
        if len(surfaces) > len(shown):
            out.append(f"… and {len(surfaces) - len(shown)} more")
        out.append(
            "A term above that you cannot map to a column is a DATA GAP, not an "
            "invitation to substitute a similar column."
        )

    if context.policies:
        out.append("")
        out.append(
            f"### Rules\n{len(context.policies)} domain-wide rule(s) apply. Look "
            f"them up before relying on a calculation."
        )

    coverage = context.authority.coverage
    if coverage.n_documents:
        # Says how much authority this card actually has. A sparse context
        # should be weighed differently, and hiding that would be dishonest.
        note = (
            f"Built from {coverage.n_documents} reviewed document(s), "
            f"{coverage.n_measures} measure(s)."
        )
        if not coverage.declared:
            note += " Scope and boundary not yet declared by an expert."
        out.append("")
        out.append(note)

    card = "\n".join(out)
    if len(card) > MAX_CARD_CHARS:
        card = card[:MAX_CARD_CHARS].rsplit("\n", 1)[0]
        card += "\n… domain card truncated; look up specific terms as needed."
        logger.warning(
            "domain_card_truncated",
            extra={"scope_id": context.scope_id, "chars": len(card),
                   "measures": len(context.measures)},
        )
    return card


def render_lookup(measures: list[Measure], rules: list[dict] | None = None) -> str:
    """Tier 2: everything known about the few terms a question touched.

    No ceiling: the caller chose these, and a lookup that silently drops what was
    asked for is worse than a long one.
    """
    if not measures and not rules:
        return "No domain knowledge recorded for those terms."

    out: list[str] = []
    for measure in measures:
        out.append(f"### {measure.name}")
        if measure.surface_forms:
            out.append(f"Also written: {', '.join(measure.surface_forms)}")
        for definition in measure.definitions or []:
            out.append(f"- {definition.text.strip()}   [{definition.doc_id}]")
        if not measure.definitions and measure.definition:
            out.append(f"- {measure.definition.strip()}")
        if measure.formula_latex:
            out.append(f"Formula: {measure.formula_latex.strip()}")
        if measure.unit:
            out.append(f"Unit: {measure.unit}")
        if measure.catalog_binding:
            out.append(f"Column: {measure.catalog_binding}")
        if measure.contested:
            out.append(
                "⚠ The documents DISAGREE about this term. Both readings are "
                "above; say which you used."
            )
        out.append("")

    for rule in rules or []:
        statement = rule.get("statement") or ""
        condition, consequence = rule.get("condition"), rule.get("consequence")
        out.append("### Rule")
        if statement:
            out.append(statement.strip())
        if condition and consequence:
            out.append(f"IF {condition.strip()} THEN {consequence.strip()}")
        out.append("")

    return "\n".join(out).strip()