Download tests/test_eval.py from DBax127/nexa: direct link, hf CLI and curl.
- Browser
- Download file 20.2 kB
-
https://huggingface.co/DBax127/nexa/resolve/main/tests/test_eval.py
- Command line
-
hf download hf://DBax127/nexa/tests/test_eval.py
-
curl -L -o test_eval.py https://huggingface.co/DBax127/nexa/resolve/main/tests/test_eval.py
20.2 kB
| """The measurement of `nexa check`, without a model in the room. | |
| Everything here is the part of `nexa eval --check` that does not generate: the | |
| file set it expands, the names it looks for in an answer, the argument | |
| combinations it refuses, and the report it prints from rows. The generation is | |
| one call and is covered where ask covers it; what is new is the bookkeeping | |
| around it, and bookkeeping is where a measurement goes quietly wrong. | |
| """ | |
| import io | |
| import json | |
| import os | |
| import pytest | |
| import eval as evalmod | |
| import subject | |
| ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) | |
| # --------------------------------------------------------- declared names | |
| def test_python_names_include_methods(): | |
| body = ("class Loader(object):\n" | |
| " def parse_row(self, row):\n" | |
| " return row\n" | |
| "\n" | |
| "def main():\n" | |
| " pass\n") | |
| assert evalmod.declared_names(body, "python") == ["Loader", "main", "parse_row"] | |
| def test_names_skip_locals_and_imported_bindings(): | |
| """`row` is a parameter and `csv` is somebody else's name. | |
| Counting them would score any answer that mentions CSV parsing as a review | |
| of this file, which is the one thing this number exists to tell apart. | |
| """ | |
| body = ("import csv\n" | |
| "\n" | |
| "def load_prices(handle):\n" | |
| " for row in csv.reader(handle):\n" | |
| " yield row\n") | |
| assert evalmod.declared_names(body, "odoo") == ["load_prices"] | |
| def test_gather_takes_files_as_given(): | |
| paths, skipped = evalmod.gather(evalmod.DEFAULT_SUBJECTS, 10) | |
| assert paths == evalmod.DEFAULT_SUBJECTS | |
| assert skipped == 0 | |
| def test_gather_walks_a_directory_and_skips_the_noise(tmp_path): | |
| (tmp_path / "src").mkdir() | |
| (tmp_path / "src" / "a.ts").write_text("export const a = 1;\n", encoding="utf-8") | |
| (tmp_path / "src" / "b.py").write_text("x = 1\n", encoding="utf-8") | |
| (tmp_path / "src" / "views.xml").write_text("<odoo/>\n", encoding="utf-8") | |
| (tmp_path / "src" / "notes.md").write_text("# no\n", encoding="utf-8") | |
| (tmp_path / "node_modules").mkdir() | |
| (tmp_path / "node_modules" / "dep.js").write_text("//\n", encoding="utf-8") | |
| paths, skipped = evalmod.gather([str(tmp_path)], 10) | |
| assert [os.path.basename(p) for p in paths] == ["b.py", "views.xml"], \ | |
| "the .ts is not reviewable and is not walked into the set" | |
| assert skipped == 0 | |
| def test_gather_caps_the_walk_and_reports_the_remainder(tmp_path): | |
| """Every file costs a generation, so a tree of 400 is not a default.""" | |
| for n in range(6): | |
| (tmp_path / "f{0}.py".format(n)).write_text("x = 1\n", encoding="utf-8") | |
| paths, skipped = evalmod.gather([str(tmp_path)], 2) | |
| assert len(paths) == 2 | |
| assert skipped == 4 | |
| def test_gather_says_which_path_is_missing(tmp_path): | |
| with pytest.raises(RuntimeError) as caught: | |
| evalmod.gather([str(tmp_path / "nope.py")], 4) | |
| assert "nope.py" in str(caught.value) | |
| # ------------------------------------------------------- argument refusals | |
| def test_check_and_pipeline_are_different_measurements(capsys): | |
| assert evalmod.main(["--check", "--pipeline"]) == 2 | |
| assert "pick one" in capsys.readouterr().out | |
| def test_files_without_check_are_refused(capsys): | |
| """Otherwise a mistyped `nexa eval subjects/` would spend an hour answering | |
| the default question set and print a report about something else.""" | |
| assert evalmod.main(["subjects/"]) == 2 | |
| assert "only reviewed with --check" in capsys.readouterr().out | |
| def test_a_missing_path_is_a_usage_error_not_a_crash(capsys, tmp_path): | |
| assert evalmod.main(["--check", str(tmp_path / "gone.py")]) == 2 | |
| assert "no such file" in capsys.readouterr().out | |
| # ------------------------------------------------------------- the report | |
| def plan_symbols(): | |
| """How many symbols the truncated subject has, asked rather than pinned.""" | |
| plan = subject.plan(evalmod.DEFAULT_SUBJECTS[1]) | |
| return len(plan["shown_names"]) + len(plan["omitted_names"]) | |
| def rows_for(verdicts, **over): | |
| """Synthetic rows for one file, one per verdict given.""" | |
| plan = subject.plan(evalmod.DEFAULT_SUBJECTS[1]) # the truncated one | |
| row = { | |
| "seconds": 20.0, "tokens_per_second": 35.0, "rules_fired": [], | |
| "layers_with_findings": [], "retrieved": ["odoo.onchange-is-not-validation"], | |
| "stack": plan["stack"], "routed_by": plan["routed_by"], "chars": plan["chars"], | |
| "dropped_chars": plan["dropped_chars"], "facts": len(plan["facts"]), | |
| "context_verdict": "WARN", "context_checked": 2, | |
| "coverage": plan["coverage"], | |
| "symbols_shown": len(plan["shown_names"]), | |
| "symbols_total": len(plan["shown_names"]) + len(plan["omitted_names"]), | |
| "file_findings": sorted({hit.rule for hit in plan["findings"]}), | |
| "file_finding_count": sum(1 for h in plan["findings"] if not h.suppressed), | |
| "file_suppressed": plan["suppressed"], | |
| "cited": ["odoo.onchange-is-not-validation"], | |
| "cited_beyond_checkers": ["odoo.onchange-is-not-validation"], | |
| "names_declared": 19, "names_mentioned": 0, "answer_chars": 400, | |
| "path": plan["path"], | |
| } | |
| row.update(over) | |
| out = [] | |
| for verdict in verdicts: | |
| one = dict(row) | |
| one["verdict"] = verdict | |
| out.append(one) | |
| return out | |
| def test_the_report_says_what_never_reached_the_model(capsys): | |
| evalmod.summarise_check(rows_for(["PASS", "PASS"]), 2) | |
| out = capsys.readouterr().out | |
| assert "WHAT REACHED THE MODEL" in out | |
| assert "7929 chars" in out | |
| assert "of {0} symbols".format(plan_symbols()) in out | |
| assert "did not fit in the 6000-character budget" in out | |
| assert "chosen by symbol" in out | |
| def test_the_report_separates_an_unstable_verdict_from_a_stable_one(capsys): | |
| evalmod.summarise_check(rows_for(["PASS", "WARN"]), 2) | |
| out = capsys.readouterr().out | |
| assert "0/1 files gave the same verdict every run" in out | |
| assert "PASS, WARN" in out | |
| evalmod.summarise_check(rows_for(["WARN", "WARN"]), 2) | |
| assert "1/1 files gave the same verdict every run" in capsys.readouterr().out | |
| def test_the_report_counts_an_answer_that_names_nothing_in_the_file(capsys): | |
| evalmod.summarise_check(rows_for(["PASS"]), 1) | |
| out = capsys.readouterr().out | |
| assert "IS THE ANSWER ABOUT THIS FILE?" in out | |
| assert "0 of 19 names mentioned" in out | |
| def test_the_report_names_the_layer_ask_cannot_run(capsys): | |
| evalmod.summarise_check(rows_for(["PASS"]), 1) | |
| out = capsys.readouterr().out | |
| assert "CONTEXT -- the layer `ask` can never run" in out | |
| assert "1/1 file(s) reached a contextual verdict" in out | |
| assert "fact(s) read in all" in out, "reading the tree is not the same as checking it" | |
| evalmod.summarise_check(rows_for(["PASS"], context_verdict="UNKNOWN"), 1) | |
| assert "0/1 file(s) reached a contextual verdict" in capsys.readouterr().out | |
| def test_a_failed_run_does_not_describe_the_file(capsys): | |
| """An ERROR row carries no stack and no size. | |
| Letting one stand for the file printed a 7,929-character module as 0 | |
| characters and its stack as "?", which reads as a finding about the file | |
| rather than what it is: one run that never reached the model. | |
| """ | |
| failed = rows_for(["ERROR"], chars=0, dropped_chars=0, stack="?", | |
| routed_by="?", error="connection refused") | |
| evalmod.summarise_check(failed + rows_for(["PASS"]), 2) | |
| out = capsys.readouterr().out | |
| assert "7929 chars" in out | |
| assert "odoo extension" in out | |
| assert "ERRORS (1)" in out | |
| def test_one_failed_run_does_not_become_a_second_file(capsys): | |
| """Stability counts files, and an ERROR row says its stack is "?". | |
| Grouped by that, one file with one failed run counted twice and the | |
| denominator came out larger than the number of files reviewed. | |
| """ | |
| failed = rows_for(["ERROR"], chars=0, stack="?", routed_by="?", error="refused") | |
| evalmod.summarise_check(failed + rows_for(["WARN", "WARN"]), 3) | |
| assert "0/1 files gave the same verdict every run" in capsys.readouterr().out | |
| def test_the_report_counts_what_the_checkers_found_in_the_file(capsys): | |
| """The number `check` did not have: findings established without a model. | |
| Twenty-five reviews reported none of the nine the corpus finds in these | |
| files on its own, because every layer ran over the answer. This section is | |
| the other half, and it does not depend on the 7B saying anything. | |
| """ | |
| evalmod.summarise_check(rows_for(["PASS"]), 1) | |
| out = capsys.readouterr().out | |
| assert "FINDINGS IN THE FILE, BEFORE THE MODEL RAN" in out | |
| assert "stored-compute-needs-complete-depends" in out | |
| assert "1/1 file(s) had something found in them this way" in out | |
| def test_the_report_says_what_the_model_added(capsys): | |
| """The question behind the whole generative half: is it worth its 4 seconds? | |
| The checkers already say what is wrong with the file. A rule the model | |
| reached for that no checker fired on is the cheapest available proxy for | |
| something the deterministic pass could not have produced. | |
| """ | |
| rows = rows_for(["PASS"]) | |
| evalmod.summarise_check(rows, 1) | |
| out = capsys.readouterr().out | |
| assert "WHAT THE MODEL ADDED" in out | |
| # Derived from the subject rather than written out. As a literal this said | |
| # 6, and narrowing a checker that was over-firing then read as a broken | |
| # test instead of as the fix it was. | |
| assert "beyond the {0} the checkers found".format( | |
| rows[0]["file_finding_count"]) in out | |
| assert "odoo.onchange-is-not-validation" in out | |
| def test_the_report_says_when_the_model_added_nothing(capsys): | |
| evalmod.summarise_check(rows_for(["PASS"], cited_beyond_checkers=[]), 1) | |
| out = capsys.readouterr().out | |
| assert "none, in any run" in out | |
| # ------------------------------------------------ the retrieval comparison | |
| def comparison_rows(whole, compact, fired=4): | |
| return [{"path": "subjects/a.py", "stack": "odoo", "fired": fired, | |
| "whole_prompt": whole, "compact": compact, | |
| "prompt_chars": 9000, "query_chars": 1200}] | |
| def test_the_comparison_is_scored_on_what_the_checkers_proved(capsys): | |
| """No model and no opinion: the checkers have already established which | |
| rules apply to the file, so a query is better if more of them come back.""" | |
| evalmod.summarise_comparison(comparison_rows(2, 4)) | |
| out = capsys.readouterr().out | |
| assert "whole prompt 2/4" in out | |
| assert "compact query 4/4" in out | |
| assert "The compact query wins" in out | |
| def test_the_comparison_says_so_when_the_new_idea_is_worse(capsys): | |
| """Written before the run, so the result could not be a foregone conclusion. | |
| A compact description of a file sounds better than nine thousand characters | |
| of source. Whether it retrieves better is a different question and this is | |
| the one that answers it. | |
| """ | |
| evalmod.summarise_comparison(comparison_rows(4, 1)) | |
| assert "The whole prompt wins" in capsys.readouterr().out | |
| def test_a_draw_is_reported_as_a_draw(capsys): | |
| evalmod.summarise_comparison(comparison_rows(3, 3)) | |
| assert "A draw on this set" in capsys.readouterr().out | |
| def test_nothing_to_score_is_not_a_result(capsys): | |
| assert evalmod.summarise_comparison([{"path": "a.py", "fired": 0}]) == 2 | |
| assert "nothing to score" in capsys.readouterr().out | |
| def test_the_report_counts_live_findings_and_says_so_about_the_rest(capsys): | |
| """file_findings and file_finding_count have to count the same things. | |
| One counted the ids of live findings and the other the length of every hit | |
| including suppressed ones, which made a file with a suppressed hit report | |
| more findings than it listed. | |
| """ | |
| evalmod.summarise_check(rows_for(["PASS"], file_finding_count=2, | |
| file_findings=["a.b", "c.d"], | |
| file_suppressed=3), 1) | |
| out = capsys.readouterr().out | |
| assert " 2 finding(s)" in out | |
| assert "3 more sat on lines another tool had already flagged" in out | |
| def test_a_review_counts_findings_the_way_check_does(fake_ollama, seeded_index): | |
| """An advisory finding is printed apart by `nexa check` and counted in | |
| nothing. This eval counted it anyway, and reported 13 findings on the | |
| subjects where fieldtest reports 10 and 3 advisory.""" | |
| fake_ollama(["The file is fine as it is."]) | |
| path = evalmod.DEFAULT_SUBJECTS[0] | |
| hits = [h for h in subject.plan(path)["findings"] if not h.suppressed] | |
| live = [h for h in hits if not h.advisory] | |
| advisory = [h for h in hits if h.advisory] | |
| assert advisory, "this subject no longer has an advisory finding to count" | |
| row = evalmod.quiet_check(path) | |
| assert row["file_finding_count"] == len(live) | |
| assert row["file_findings"] == sorted({h.rule for h in live}) | |
| assert row["file_advisory"] == len(advisory) | |
| def test_the_report_says_what_it_left_out_as_advisory(capsys): | |
| evalmod.summarise_check(rows_for(["PASS"], file_advisory=2), 1) | |
| assert "2 more came from advisory rules" in capsys.readouterr().out | |
| # ----------------------------------------------------- the confidence interval | |
| def test_wilson_stays_inside_zero_and_one_at_tiny_n(): | |
| """The reason it is Wilson and not mean +/- 1.96*sigma. The normal | |
| approximation on 2 of 2 produces bounds outside [0, 1] and a width that | |
| claims more precision than two samples can carry.""" | |
| low, high = evalmod.wilson(2, 2) | |
| assert 0.0 <= low <= high <= 1.0 | |
| assert low < 1.0, "two successes is not certainty" | |
| def test_wilson_on_no_samples_is_the_whole_interval(): | |
| """Nothing observed constrains nothing. Reporting 0% here would be a claim | |
| about a rate nobody has measured.""" | |
| assert evalmod.wilson(0, 0) == (0.0, 1.0) | |
| def test_wilson_narrows_as_the_sample_grows(): | |
| """The property the published rescue rate depends on: 12 of 22 is an | |
| interval and 1200 of 2200 is nearly a number.""" | |
| small = evalmod.wilson(12, 22) | |
| large = evalmod.wilson(1200, 2200) | |
| assert (large[1] - large[0]) < (small[1] - small[0]) | |
| for low, high in (small, large): | |
| assert low < 12.0 / 22 < high | |
| def test_wilson_brackets_the_published_rescue_rate(): | |
| """README quotes 11 of 21, 95% CI 32-72%. If this ever moves, the document | |
| is wrong and nothing else would say so.""" | |
| low, high = evalmod.wilson(11, 21) | |
| assert round(100 * low) == 32 | |
| assert round(100 * high) == 72 | |
| # ------------------------------------------------------ the pipeline summary | |
| def pipeline_row(**overrides): | |
| row = {"stack": "odoo", "question": "a question", "verdict": "PASS", | |
| "seconds": 1.0, "tokens_per_second": 10.0, "escalated": False, | |
| "rescued": False, "rules_fired": [], "layers_with_findings": [], | |
| "retrieved": []} | |
| row.update(overrides) | |
| return row | |
| def test_the_pipeline_summary_reports_escalation_as_an_interval(capsys): | |
| """A rescue rate off a handful of escalations is not a percentage, and the | |
| report has to say which of the two it is holding.""" | |
| rows = [pipeline_row(escalated=True, rescued=True) for _ in range(3)] | |
| rows += [pipeline_row(escalated=True, rescued=False)] | |
| rows += [pipeline_row() for _ in range(4)] | |
| evalmod.summarise(rows, 1, pipeline_mode=True) | |
| out = capsys.readouterr().out | |
| assert "ESCALATION" in out | |
| assert "triggered 4/8" in out | |
| assert "rescued 3/4" in out | |
| assert "95% CI" in out | |
| def test_no_escalation_reports_no_rescue_rate(capsys): | |
| """Zero of zero rescued is not 0%, and printing it would invite reading it | |
| as one.""" | |
| evalmod.summarise([pipeline_row() for _ in range(4)], 1, pipeline_mode=True) | |
| out = capsys.readouterr().out | |
| assert "triggered 0/4" in out | |
| assert "rescued" not in out | |
| def test_the_ask_summary_names_rules_that_never_surfaced(capsys): | |
| """Scoped to the stack that actually ran, and the never-retrieved list is | |
| what the summary exists to print: a rule nothing surfaced is a rule no | |
| question in the set could reach.""" | |
| rows = [pipeline_row(stack="odoo", | |
| retrieved=["odoo.onchange-is-not-validation"], | |
| rules_fired=["odoo.onchange-is-not-validation"], | |
| layers_with_findings=["domain"])] | |
| evalmod.summarise(rows, 1, pipeline_mode=False) | |
| out = capsys.readouterr().out | |
| assert "RETRIEVAL (odoo)" in out | |
| assert "never odoo." in out, "the rules no question in the set reached" | |
| assert "1x odoo.onchange-is-not-validation" in out | |
| def test_the_ask_summary_says_so_when_no_checker_fired(capsys): | |
| """Silence from the checkers has two readings and the report refuses to | |
| pick one.""" | |
| evalmod.summarise([pipeline_row(stack="odoo")], 1, pipeline_mode=False) | |
| out = capsys.readouterr().out | |
| assert "the answers were clean or the checkers are asleep" in out | |
| # ----------------------------------------------------- the retrieval summary | |
| def retrieval_row(rid, **overrides): | |
| row = {"id": rid, "stack": rid.split(".")[0], "probe": "how do I do the thing", | |
| "rank": 1, "rank_unpinned": 1, "pinned": False, "instead": []} | |
| row.update(overrides) | |
| return row | |
| def test_a_rule_that_does_not_answer_its_own_probe_is_reported(capsys): | |
| """Unreachable by any question at all, which no verdict would ever show.""" | |
| rows = [retrieval_row("odoo.sudo-bypasses-record-rules"), | |
| retrieval_row("odoo.no-monkey-patching", rank=None, | |
| rank_unpinned=None, | |
| instead=["odoo.ondelete-is-a-decision"])] | |
| code = evalmod.summarise_retrieval(rows, k=6) | |
| out = capsys.readouterr().out | |
| assert code == 1 | |
| assert "NOT RETRIEVED BY THEIR OWN PROBE (1)" in out | |
| assert "came back instead: odoo.ondelete-is-a-decision" in out | |
| def test_a_rule_with_no_probe_is_unmeasured_not_unreachable(capsys): | |
| """Two different things, and collapsing them would report a rule nobody | |
| wrote a question for as a rule nobody can find.""" | |
| rows = [retrieval_row("odoo.sudo-bypasses-record-rules"), | |
| retrieval_row("odoo.no-monkey-patching", probe=None, rank=None, | |
| rank_unpinned=None)] | |
| code = evalmod.summarise_retrieval(rows, k=6) | |
| out = capsys.readouterr().out | |
| assert code == 1 | |
| assert "NO PROBE (1) -- findability not measured" in out | |
| assert "NOT RETRIEVED" not in out | |
| def test_a_pinned_rule_is_excluded_from_the_ranking(capsys): | |
| """It is prepended regardless of similarity, so counting it as ranked | |
| first would be scoring the pin rather than the wording.""" | |
| rows = [retrieval_row("odoo.sudo-bypasses-record-rules"), | |
| retrieval_row("odoo.check-constraint-cannot-cross-tables", | |
| pinned=True, rank=1, rank_unpinned=None)] | |
| assert evalmod.summarise_retrieval(rows, k=6) == 0 | |
| out = capsys.readouterr().out | |
| assert "ranked first 1/1" in out | |
| assert "always retrieved, so not a measurement" in out | |
| def test_a_k_that_cannot_fail_says_so(capsys): | |
| """At k >= the size of the stack, every rule comes back and the test has | |
| measured nothing. Printing a green tick for that is the failure mode.""" | |
| rows = [retrieval_row("odoo.onchange-is-not-validation")] | |
| evalmod.summarise_retrieval(rows, k=40) | |
| assert "nothing can miss" in capsys.readouterr().out | |
| def test_the_retrieval_artifact_says_its_depth_and_its_corpus(tmp_path, monkeypatch): | |
| """Bare rows could be checked against nothing: a rank means nothing without | |
| the k it was taken at, and a reworded rule keeps its id. figures.py reads | |
| both from here.""" | |
| monkeypatch.setattr(evalmod, "retrievability", lambda stack, k: [ | |
| retrieval_row("odoo.onchange-is-not-validation")]) | |
| out = tmp_path / "retrieval.json" | |
| assert evalmod.main(["--retrieval", "--top-k", "4", "--json", str(out)]) == 0 | |
| art = json.loads(out.read_text(encoding="utf-8")) | |
| assert art["k"] == 4 | |
| assert art["corpus_sha256"] == evalmod.evidence.corpus_sha256() | |
| assert [r["id"] for r in art["rows"]] == ["odoo.onchange-is-not-validation"] | |