Spaces:
Sleeping
Sleeping
File size: 10,478 Bytes
b2931f4 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 | """Day-4 evaluation set β the questions that make the system *provable*.
Four tiers, each probing a distinct failure mode of an agentic RAG system:
factual β sql route. One exact top-level figure. We assert the number
appears in the answer. Ground truth is read from our own
`financial_facts` table (correct-by-construction), so this tier
is really a regression check: does the agent route to sql and
report the figure *faithfully*, or hallucinate / mis-label it
(the exact bug that survived Day 2)?
narrative β vector route. Qualitative content. No numeric ground truth;
graded by the LLM judge on faithfulness + answer relevance, and
by a deterministic check that the answer actually cites [N].
multihop β both + calculator. A figure that must be *computed* (YoY growth,
margin) or a cross-company comparison. Tests retrieval + tool use
+ arithmetic end to end.
honesty β unanswerable from the corpus (a company/year we don't have, or a
forward-looking number). The agent must DECLINE, not fabricate β
the single most important behaviour for a finance assistant.
Ground truth for the numeric tiers is a `gt` callable resolved lazily against
DuckDB, so this module imports without touching the DB and the numbers can
never drift from what the agent can actually query.
"""
from __future__ import annotations
from collections.abc import Callable
from dataclasses import dataclass
from finrag.ingestion.facts import query
# ββ Tiers ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
FACTUAL = "factual"
NARRATIVE = "narrative"
MULTIHOP = "multihop"
HONESTY = "honesty"
@dataclass(frozen=True)
class EvalCase:
id: str
tier: str
question: str
# Soft expectation β the plan node's route. We record match/mismatch but a
# mismatch isn't a hard failure (the agent can still answer correctly via a
# different path).
route_expected: str | None = None
# Numeric ground truth, resolved lazily from the DB at run time.
gt: Callable[[], float] | None = None
gt_kind: str = "currency" # currency | per_share | percent
# Relative tolerance for currency/per_share; for percent it's the absolute
# percentage-point floor (combined with a 5% relative band in the matcher).
tol: float = 0.015
# For name-style answers ("which companyβ¦") and any case where a substring
# must appear (e.g. the right company name).
expect_substring: str | None = None
# Honesty tier: the agent must refuse / say it's not in the corpus.
must_decline: bool = False
# ββ Ground-truth helper ββββββββββββββββββββββββββββββββββββββββββββββββββ
def _v(ticker: str, line_item: str, fy: int, unit: str = "USD") -> float:
"""The single authoritative value for a (ticker, line_item, year) from
financial_facts β the same table sql_query reads. LIMIT 1 because each of
our three filers resolves to one surviving GAAP concept per annual period
(see facts.py dedup); ordering by value desc is a stable tie-break."""
rows = query(
"""
SELECT value FROM financial_facts
WHERE ticker = ? AND line_item = ? AND fiscal_year = ?
AND fiscal_period = 'FY' AND unit = ?
ORDER BY value DESC LIMIT 1
""",
[ticker, line_item, fy, unit],
)
if not rows:
raise LookupError(f"no fact for {ticker} {line_item} FY{fy} ({unit})")
return float(rows[0]["value"])
def _yoy(ticker: str, line_item: str, y0: int, y1: int) -> float:
"""Year-over-year percent growth from y0 to y1."""
a, b = _v(ticker, line_item, y0), _v(ticker, line_item, y1)
return (b - a) / a * 100.0
def _margin(ticker: str, part: str, whole: str, fy: int) -> float:
"""A margin in percent (e.g. gross_profit / revenue)."""
return _v(ticker, part, fy) / _v(ticker, whole, fy) * 100.0
# ββ The set ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
CASES: list[EvalCase] = [
# βββββββββββ FACTUAL (sql route, exact figure) βββββββββββ
EvalCase("f01", FACTUAL, "What was Apple's net income in fiscal 2023?",
route_expected="sql", gt=lambda: _v("AAPL", "net_income", 2023)),
EvalCase("f02", FACTUAL, "What was Apple's total revenue in fiscal 2024?",
route_expected="sql", gt=lambda: _v("AAPL", "revenue", 2024)),
EvalCase("f03", FACTUAL, "How much did Tesla spend on research and development in fiscal 2023?",
route_expected="sql", gt=lambda: _v("TSLA", "rd_expense", 2023)),
EvalCase("f04", FACTUAL, "What were JPMorgan's total assets at the end of fiscal 2023?",
route_expected="sql", gt=lambda: _v("JPM", "total_assets", 2023)),
EvalCase("f05", FACTUAL, "What was Apple's operating income in fiscal 2022?",
route_expected="sql", gt=lambda: _v("AAPL", "operating_income", 2022)),
EvalCase("f06", FACTUAL, "What was Tesla's total revenue in fiscal 2024?",
route_expected="sql", gt=lambda: _v("TSLA", "revenue", 2024)),
EvalCase("f07", FACTUAL, "What was JPMorgan's net income in fiscal 2024?",
route_expected="sql", gt=lambda: _v("JPM", "net_income", 2024)),
EvalCase("f08", FACTUAL, "What was Apple's gross profit in fiscal 2023?",
route_expected="sql", gt=lambda: _v("AAPL", "gross_profit", 2023)),
EvalCase("f09", FACTUAL, "What was Tesla's diluted earnings per share in fiscal 2023?",
route_expected="sql", gt=lambda: _v("TSLA", "eps_diluted", 2023, "USD/shares"),
gt_kind="per_share"),
EvalCase("f10", FACTUAL, "What was JPMorgan's net interest income in fiscal 2024?",
route_expected="sql", gt=lambda: _v("JPM", "net_interest_income", 2024)),
# βββββββββββ NARRATIVE (vector route, faithfulness + citation) βββββββββββ
EvalCase("n01", NARRATIVE, "How does Apple describe the risks to its supply chain in its 10-K?",
route_expected="vector"),
EvalCase("n02", NARRATIVE, "What does Tesla say about competition in the electric-vehicle market?",
route_expected="vector"),
EvalCase("n03", NARRATIVE, "How does JPMorgan describe credit risk in its filing?",
route_expected="vector"),
EvalCase("n04", NARRATIVE, "How does Apple characterize its Services business and what drives its growth?",
route_expected="vector"),
EvalCase("n05", NARRATIVE, "What risks does Tesla cite related to its dependence on key personnel?",
route_expected="vector"),
EvalCase("n06", NARRATIVE, "How does Apple describe foreign-currency exchange-rate risk?",
route_expected="vector"),
EvalCase("n07", NARRATIVE, "What does Tesla say about risks in ramping production and manufacturing?",
route_expected="vector"),
EvalCase("n08", NARRATIVE, "How does JPMorgan describe the regulatory and capital requirements it faces?",
route_expected="vector"),
# βββββββββββ MULTIHOP (sql + calculator; route left soft) βββββββββββ
# route_expected is None: each of these is answerable from structured facts
# plus arithmetic, so the router legitimately picks `sql` over `both`. The
# tier tests multi-step reasoning (fetch figure(s) β compute), not the route.
EvalCase("m01", MULTIHOP, "By what percentage did Apple's net income change from fiscal 2022 to fiscal 2023?",
gt=lambda: _yoy("AAPL", "net_income", 2022, 2023), gt_kind="percent"),
EvalCase("m02", MULTIHOP, "By what percentage did Tesla's revenue grow from fiscal 2023 to fiscal 2024?",
gt=lambda: _yoy("TSLA", "revenue", 2023, 2024), gt_kind="percent"),
EvalCase("m03", MULTIHOP, "What was Apple's gross margin in fiscal 2023?",
gt=lambda: _margin("AAPL", "gross_profit", "revenue", 2023), gt_kind="percent"),
EvalCase("m04", MULTIHOP, "What was Tesla's net profit margin in fiscal 2023?",
gt=lambda: _margin("TSLA", "net_income", "revenue", 2023), gt_kind="percent"),
EvalCase("m05", MULTIHOP, "What was Apple's operating margin in fiscal 2024?",
gt=lambda: _margin("AAPL", "operating_income", "revenue", 2024), gt_kind="percent"),
EvalCase("m06", MULTIHOP, "Which of the three companies had the highest net income in fiscal 2023?",
expect_substring="Apple"),
EvalCase("m07", MULTIHOP, "By what percentage did JPMorgan's net income change from fiscal 2023 to fiscal 2024?",
gt=lambda: _yoy("JPM", "net_income", 2023, 2024), gt_kind="percent"),
# βββββββββββ HONESTY (must decline β outside the corpus) βββββββββββ
EvalCase("h01", HONESTY, "What was Microsoft's net income in fiscal 2023?",
must_decline=True),
EvalCase("h02", HONESTY, "What was Apple's total revenue in fiscal 2019?",
must_decline=True), # outside our FY2022β2024 span
EvalCase("h03", HONESTY, "How many employees does Amazon have according to its 10-K?",
must_decline=True), # company not in corpus
EvalCase("h04", HONESTY, "What will Tesla's revenue be in fiscal 2026?",
must_decline=True), # forward-looking, not in any filing
EvalCase("h05", HONESTY, "What was Google's operating margin in fiscal 2023?",
must_decline=True), # company not in corpus
]
def cases_for(tier: str | None = None) -> list[EvalCase]:
return [c for c in CASES if tier is None or c.tier == tier]
def one_per_tier() -> list[EvalCase]:
"""A 4-case smoke subset β first case of each tier. Used to validate the
harness end to end before spending on a full run, and as the Gemini A/B
subset that fits the free-tier daily quota."""
out: list[EvalCase] = []
for tier in (FACTUAL, NARRATIVE, MULTIHOP, HONESTY):
out.append(cases_for(tier)[0])
return out
|