nexa / tests /test_eval.py
DBax127's picture
check: the eval rows re-measured on the subjects left, and checked
3ac608a
Raw History Blame Contribute Delete
20.2 kB
"""The measurement of `nexa check`, without a model in the room.
Everything here is the part of `nexa eval --check` that does not generate: the
file set it expands, the names it looks for in an answer, the argument
combinations it refuses, and the report it prints from rows. The generation is
one call and is covered where ask covers it; what is new is the bookkeeping
around it, and bookkeeping is where a measurement goes quietly wrong.
"""
import io
import json
import os
import pytest
import eval as evalmod
import subject
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
# --------------------------------------------------------- declared names
def test_python_names_include_methods():
body = ("class Loader(object):\n"
" def parse_row(self, row):\n"
" return row\n"
"\n"
"def main():\n"
" pass\n")
assert evalmod.declared_names(body, "python") == ["Loader", "main", "parse_row"]
def test_names_skip_locals_and_imported_bindings():
"""`row` is a parameter and `csv` is somebody else's name.
Counting them would score any answer that mentions CSV parsing as a review
of this file, which is the one thing this number exists to tell apart.
"""
body = ("import csv\n"
"\n"
"def load_prices(handle):\n"
" for row in csv.reader(handle):\n"
" yield row\n")
assert evalmod.declared_names(body, "odoo") == ["load_prices"]
def test_gather_takes_files_as_given():
paths, skipped = evalmod.gather(evalmod.DEFAULT_SUBJECTS, 10)
assert paths == evalmod.DEFAULT_SUBJECTS
assert skipped == 0
def test_gather_walks_a_directory_and_skips_the_noise(tmp_path):
(tmp_path / "src").mkdir()
(tmp_path / "src" / "a.ts").write_text("export const a = 1;\n", encoding="utf-8")
(tmp_path / "src" / "b.py").write_text("x = 1\n", encoding="utf-8")
(tmp_path / "src" / "views.xml").write_text("<odoo/>\n", encoding="utf-8")
(tmp_path / "src" / "notes.md").write_text("# no\n", encoding="utf-8")
(tmp_path / "node_modules").mkdir()
(tmp_path / "node_modules" / "dep.js").write_text("//\n", encoding="utf-8")
paths, skipped = evalmod.gather([str(tmp_path)], 10)
assert [os.path.basename(p) for p in paths] == ["b.py", "views.xml"], \
"the .ts is not reviewable and is not walked into the set"
assert skipped == 0
def test_gather_caps_the_walk_and_reports_the_remainder(tmp_path):
"""Every file costs a generation, so a tree of 400 is not a default."""
for n in range(6):
(tmp_path / "f{0}.py".format(n)).write_text("x = 1\n", encoding="utf-8")
paths, skipped = evalmod.gather([str(tmp_path)], 2)
assert len(paths) == 2
assert skipped == 4
def test_gather_says_which_path_is_missing(tmp_path):
with pytest.raises(RuntimeError) as caught:
evalmod.gather([str(tmp_path / "nope.py")], 4)
assert "nope.py" in str(caught.value)
# ------------------------------------------------------- argument refusals
def test_check_and_pipeline_are_different_measurements(capsys):
assert evalmod.main(["--check", "--pipeline"]) == 2
assert "pick one" in capsys.readouterr().out
def test_files_without_check_are_refused(capsys):
"""Otherwise a mistyped `nexa eval subjects/` would spend an hour answering
the default question set and print a report about something else."""
assert evalmod.main(["subjects/"]) == 2
assert "only reviewed with --check" in capsys.readouterr().out
def test_a_missing_path_is_a_usage_error_not_a_crash(capsys, tmp_path):
assert evalmod.main(["--check", str(tmp_path / "gone.py")]) == 2
assert "no such file" in capsys.readouterr().out
# ------------------------------------------------------------- the report
def plan_symbols():
"""How many symbols the truncated subject has, asked rather than pinned."""
plan = subject.plan(evalmod.DEFAULT_SUBJECTS[1])
return len(plan["shown_names"]) + len(plan["omitted_names"])
def rows_for(verdicts, **over):
"""Synthetic rows for one file, one per verdict given."""
plan = subject.plan(evalmod.DEFAULT_SUBJECTS[1]) # the truncated one
row = {
"seconds": 20.0, "tokens_per_second": 35.0, "rules_fired": [],
"layers_with_findings": [], "retrieved": ["odoo.onchange-is-not-validation"],
"stack": plan["stack"], "routed_by": plan["routed_by"], "chars": plan["chars"],
"dropped_chars": plan["dropped_chars"], "facts": len(plan["facts"]),
"context_verdict": "WARN", "context_checked": 2,
"coverage": plan["coverage"],
"symbols_shown": len(plan["shown_names"]),
"symbols_total": len(plan["shown_names"]) + len(plan["omitted_names"]),
"file_findings": sorted({hit.rule for hit in plan["findings"]}),
"file_finding_count": sum(1 for h in plan["findings"] if not h.suppressed),
"file_suppressed": plan["suppressed"],
"cited": ["odoo.onchange-is-not-validation"],
"cited_beyond_checkers": ["odoo.onchange-is-not-validation"],
"names_declared": 19, "names_mentioned": 0, "answer_chars": 400,
"path": plan["path"],
}
row.update(over)
out = []
for verdict in verdicts:
one = dict(row)
one["verdict"] = verdict
out.append(one)
return out
def test_the_report_says_what_never_reached_the_model(capsys):
evalmod.summarise_check(rows_for(["PASS", "PASS"]), 2)
out = capsys.readouterr().out
assert "WHAT REACHED THE MODEL" in out
assert "7929 chars" in out
assert "of {0} symbols".format(plan_symbols()) in out
assert "did not fit in the 6000-character budget" in out
assert "chosen by symbol" in out
def test_the_report_separates_an_unstable_verdict_from_a_stable_one(capsys):
evalmod.summarise_check(rows_for(["PASS", "WARN"]), 2)
out = capsys.readouterr().out
assert "0/1 files gave the same verdict every run" in out
assert "PASS, WARN" in out
evalmod.summarise_check(rows_for(["WARN", "WARN"]), 2)
assert "1/1 files gave the same verdict every run" in capsys.readouterr().out
def test_the_report_counts_an_answer_that_names_nothing_in_the_file(capsys):
evalmod.summarise_check(rows_for(["PASS"]), 1)
out = capsys.readouterr().out
assert "IS THE ANSWER ABOUT THIS FILE?" in out
assert "0 of 19 names mentioned" in out
def test_the_report_names_the_layer_ask_cannot_run(capsys):
evalmod.summarise_check(rows_for(["PASS"]), 1)
out = capsys.readouterr().out
assert "CONTEXT -- the layer `ask` can never run" in out
assert "1/1 file(s) reached a contextual verdict" in out
assert "fact(s) read in all" in out, "reading the tree is not the same as checking it"
evalmod.summarise_check(rows_for(["PASS"], context_verdict="UNKNOWN"), 1)
assert "0/1 file(s) reached a contextual verdict" in capsys.readouterr().out
def test_a_failed_run_does_not_describe_the_file(capsys):
"""An ERROR row carries no stack and no size.
Letting one stand for the file printed a 7,929-character module as 0
characters and its stack as "?", which reads as a finding about the file
rather than what it is: one run that never reached the model.
"""
failed = rows_for(["ERROR"], chars=0, dropped_chars=0, stack="?",
routed_by="?", error="connection refused")
evalmod.summarise_check(failed + rows_for(["PASS"]), 2)
out = capsys.readouterr().out
assert "7929 chars" in out
assert "odoo extension" in out
assert "ERRORS (1)" in out
def test_one_failed_run_does_not_become_a_second_file(capsys):
"""Stability counts files, and an ERROR row says its stack is "?".
Grouped by that, one file with one failed run counted twice and the
denominator came out larger than the number of files reviewed.
"""
failed = rows_for(["ERROR"], chars=0, stack="?", routed_by="?", error="refused")
evalmod.summarise_check(failed + rows_for(["WARN", "WARN"]), 3)
assert "0/1 files gave the same verdict every run" in capsys.readouterr().out
def test_the_report_counts_what_the_checkers_found_in_the_file(capsys):
"""The number `check` did not have: findings established without a model.
Twenty-five reviews reported none of the nine the corpus finds in these
files on its own, because every layer ran over the answer. This section is
the other half, and it does not depend on the 7B saying anything.
"""
evalmod.summarise_check(rows_for(["PASS"]), 1)
out = capsys.readouterr().out
assert "FINDINGS IN THE FILE, BEFORE THE MODEL RAN" in out
assert "stored-compute-needs-complete-depends" in out
assert "1/1 file(s) had something found in them this way" in out
def test_the_report_says_what_the_model_added(capsys):
"""The question behind the whole generative half: is it worth its 4 seconds?
The checkers already say what is wrong with the file. A rule the model
reached for that no checker fired on is the cheapest available proxy for
something the deterministic pass could not have produced.
"""
rows = rows_for(["PASS"])
evalmod.summarise_check(rows, 1)
out = capsys.readouterr().out
assert "WHAT THE MODEL ADDED" in out
# Derived from the subject rather than written out. As a literal this said
# 6, and narrowing a checker that was over-firing then read as a broken
# test instead of as the fix it was.
assert "beyond the {0} the checkers found".format(
rows[0]["file_finding_count"]) in out
assert "odoo.onchange-is-not-validation" in out
def test_the_report_says_when_the_model_added_nothing(capsys):
evalmod.summarise_check(rows_for(["PASS"], cited_beyond_checkers=[]), 1)
out = capsys.readouterr().out
assert "none, in any run" in out
# ------------------------------------------------ the retrieval comparison
def comparison_rows(whole, compact, fired=4):
return [{"path": "subjects/a.py", "stack": "odoo", "fired": fired,
"whole_prompt": whole, "compact": compact,
"prompt_chars": 9000, "query_chars": 1200}]
def test_the_comparison_is_scored_on_what_the_checkers_proved(capsys):
"""No model and no opinion: the checkers have already established which
rules apply to the file, so a query is better if more of them come back."""
evalmod.summarise_comparison(comparison_rows(2, 4))
out = capsys.readouterr().out
assert "whole prompt 2/4" in out
assert "compact query 4/4" in out
assert "The compact query wins" in out
def test_the_comparison_says_so_when_the_new_idea_is_worse(capsys):
"""Written before the run, so the result could not be a foregone conclusion.
A compact description of a file sounds better than nine thousand characters
of source. Whether it retrieves better is a different question and this is
the one that answers it.
"""
evalmod.summarise_comparison(comparison_rows(4, 1))
assert "The whole prompt wins" in capsys.readouterr().out
def test_a_draw_is_reported_as_a_draw(capsys):
evalmod.summarise_comparison(comparison_rows(3, 3))
assert "A draw on this set" in capsys.readouterr().out
def test_nothing_to_score_is_not_a_result(capsys):
assert evalmod.summarise_comparison([{"path": "a.py", "fired": 0}]) == 2
assert "nothing to score" in capsys.readouterr().out
def test_the_report_counts_live_findings_and_says_so_about_the_rest(capsys):
"""file_findings and file_finding_count have to count the same things.
One counted the ids of live findings and the other the length of every hit
including suppressed ones, which made a file with a suppressed hit report
more findings than it listed.
"""
evalmod.summarise_check(rows_for(["PASS"], file_finding_count=2,
file_findings=["a.b", "c.d"],
file_suppressed=3), 1)
out = capsys.readouterr().out
assert " 2 finding(s)" in out
assert "3 more sat on lines another tool had already flagged" in out
def test_a_review_counts_findings_the_way_check_does(fake_ollama, seeded_index):
"""An advisory finding is printed apart by `nexa check` and counted in
nothing. This eval counted it anyway, and reported 13 findings on the
subjects where fieldtest reports 10 and 3 advisory."""
fake_ollama(["The file is fine as it is."])
path = evalmod.DEFAULT_SUBJECTS[0]
hits = [h for h in subject.plan(path)["findings"] if not h.suppressed]
live = [h for h in hits if not h.advisory]
advisory = [h for h in hits if h.advisory]
assert advisory, "this subject no longer has an advisory finding to count"
row = evalmod.quiet_check(path)
assert row["file_finding_count"] == len(live)
assert row["file_findings"] == sorted({h.rule for h in live})
assert row["file_advisory"] == len(advisory)
def test_the_report_says_what_it_left_out_as_advisory(capsys):
evalmod.summarise_check(rows_for(["PASS"], file_advisory=2), 1)
assert "2 more came from advisory rules" in capsys.readouterr().out
# ----------------------------------------------------- the confidence interval
def test_wilson_stays_inside_zero_and_one_at_tiny_n():
"""The reason it is Wilson and not mean +/- 1.96*sigma. The normal
approximation on 2 of 2 produces bounds outside [0, 1] and a width that
claims more precision than two samples can carry."""
low, high = evalmod.wilson(2, 2)
assert 0.0 <= low <= high <= 1.0
assert low < 1.0, "two successes is not certainty"
def test_wilson_on_no_samples_is_the_whole_interval():
"""Nothing observed constrains nothing. Reporting 0% here would be a claim
about a rate nobody has measured."""
assert evalmod.wilson(0, 0) == (0.0, 1.0)
def test_wilson_narrows_as_the_sample_grows():
"""The property the published rescue rate depends on: 12 of 22 is an
interval and 1200 of 2200 is nearly a number."""
small = evalmod.wilson(12, 22)
large = evalmod.wilson(1200, 2200)
assert (large[1] - large[0]) < (small[1] - small[0])
for low, high in (small, large):
assert low < 12.0 / 22 < high
def test_wilson_brackets_the_published_rescue_rate():
"""README quotes 11 of 21, 95% CI 32-72%. If this ever moves, the document
is wrong and nothing else would say so."""
low, high = evalmod.wilson(11, 21)
assert round(100 * low) == 32
assert round(100 * high) == 72
# ------------------------------------------------------ the pipeline summary
def pipeline_row(**overrides):
row = {"stack": "odoo", "question": "a question", "verdict": "PASS",
"seconds": 1.0, "tokens_per_second": 10.0, "escalated": False,
"rescued": False, "rules_fired": [], "layers_with_findings": [],
"retrieved": []}
row.update(overrides)
return row
def test_the_pipeline_summary_reports_escalation_as_an_interval(capsys):
"""A rescue rate off a handful of escalations is not a percentage, and the
report has to say which of the two it is holding."""
rows = [pipeline_row(escalated=True, rescued=True) for _ in range(3)]
rows += [pipeline_row(escalated=True, rescued=False)]
rows += [pipeline_row() for _ in range(4)]
evalmod.summarise(rows, 1, pipeline_mode=True)
out = capsys.readouterr().out
assert "ESCALATION" in out
assert "triggered 4/8" in out
assert "rescued 3/4" in out
assert "95% CI" in out
def test_no_escalation_reports_no_rescue_rate(capsys):
"""Zero of zero rescued is not 0%, and printing it would invite reading it
as one."""
evalmod.summarise([pipeline_row() for _ in range(4)], 1, pipeline_mode=True)
out = capsys.readouterr().out
assert "triggered 0/4" in out
assert "rescued" not in out
def test_the_ask_summary_names_rules_that_never_surfaced(capsys):
"""Scoped to the stack that actually ran, and the never-retrieved list is
what the summary exists to print: a rule nothing surfaced is a rule no
question in the set could reach."""
rows = [pipeline_row(stack="odoo",
retrieved=["odoo.onchange-is-not-validation"],
rules_fired=["odoo.onchange-is-not-validation"],
layers_with_findings=["domain"])]
evalmod.summarise(rows, 1, pipeline_mode=False)
out = capsys.readouterr().out
assert "RETRIEVAL (odoo)" in out
assert "never odoo." in out, "the rules no question in the set reached"
assert "1x odoo.onchange-is-not-validation" in out
def test_the_ask_summary_says_so_when_no_checker_fired(capsys):
"""Silence from the checkers has two readings and the report refuses to
pick one."""
evalmod.summarise([pipeline_row(stack="odoo")], 1, pipeline_mode=False)
out = capsys.readouterr().out
assert "the answers were clean or the checkers are asleep" in out
# ----------------------------------------------------- the retrieval summary
def retrieval_row(rid, **overrides):
row = {"id": rid, "stack": rid.split(".")[0], "probe": "how do I do the thing",
"rank": 1, "rank_unpinned": 1, "pinned": False, "instead": []}
row.update(overrides)
return row
def test_a_rule_that_does_not_answer_its_own_probe_is_reported(capsys):
"""Unreachable by any question at all, which no verdict would ever show."""
rows = [retrieval_row("odoo.sudo-bypasses-record-rules"),
retrieval_row("odoo.no-monkey-patching", rank=None,
rank_unpinned=None,
instead=["odoo.ondelete-is-a-decision"])]
code = evalmod.summarise_retrieval(rows, k=6)
out = capsys.readouterr().out
assert code == 1
assert "NOT RETRIEVED BY THEIR OWN PROBE (1)" in out
assert "came back instead: odoo.ondelete-is-a-decision" in out
def test_a_rule_with_no_probe_is_unmeasured_not_unreachable(capsys):
"""Two different things, and collapsing them would report a rule nobody
wrote a question for as a rule nobody can find."""
rows = [retrieval_row("odoo.sudo-bypasses-record-rules"),
retrieval_row("odoo.no-monkey-patching", probe=None, rank=None,
rank_unpinned=None)]
code = evalmod.summarise_retrieval(rows, k=6)
out = capsys.readouterr().out
assert code == 1
assert "NO PROBE (1) -- findability not measured" in out
assert "NOT RETRIEVED" not in out
def test_a_pinned_rule_is_excluded_from_the_ranking(capsys):
"""It is prepended regardless of similarity, so counting it as ranked
first would be scoring the pin rather than the wording."""
rows = [retrieval_row("odoo.sudo-bypasses-record-rules"),
retrieval_row("odoo.check-constraint-cannot-cross-tables",
pinned=True, rank=1, rank_unpinned=None)]
assert evalmod.summarise_retrieval(rows, k=6) == 0
out = capsys.readouterr().out
assert "ranked first 1/1" in out
assert "always retrieved, so not a measurement" in out
def test_a_k_that_cannot_fail_says_so(capsys):
"""At k >= the size of the stack, every rule comes back and the test has
measured nothing. Printing a green tick for that is the failure mode."""
rows = [retrieval_row("odoo.onchange-is-not-validation")]
evalmod.summarise_retrieval(rows, k=40)
assert "nothing can miss" in capsys.readouterr().out
def test_the_retrieval_artifact_says_its_depth_and_its_corpus(tmp_path, monkeypatch):
"""Bare rows could be checked against nothing: a rank means nothing without
the k it was taken at, and a reworded rule keeps its id. figures.py reads
both from here."""
monkeypatch.setattr(evalmod, "retrievability", lambda stack, k: [
retrieval_row("odoo.onchange-is-not-validation")])
out = tmp_path / "retrieval.json"
assert evalmod.main(["--retrieval", "--top-k", "4", "--json", str(out)]) == 0
art = json.loads(out.read_text(encoding="utf-8"))
assert art["k"] == 4
assert art["corpus_sha256"] == evalmod.evidence.corpus_sha256()
assert [r["id"] for r in art["rows"]] == ["odoo.onchange-is-not-validation"]