"""Error-classification validation harness (Wave 1, Item 10). Deterministic rules only — no LLM, no model calls, no network. Tests 6 canonical error-classification cases and verifies the v1.1 graceful- degradation rule: where deterministic rules cannot reliably separate ``unit_error`` from ``calculation_error``, the distinction merges into the broader ``numerical_error`` bucket (Product Strategy §15.3). """ from __future__ import annotations from app.services.mastery_engine import classify_error # --------------------------------------------------------------------------- # 1. Blank / whitespace → not_attempted # --------------------------------------------------------------------------- def test_blank_answer_is_not_attempted() -> None: assert ( classify_error( question_type="numerical", student_answer="", correct_answer="340 m/s", ) == "not_attempted" ) def test_whitespace_only_answer_is_not_attempted() -> None: assert ( classify_error( question_type="numerical", student_answer=" \t \n ", correct_answer="340 m/s", ) == "not_attempted" ) # --------------------------------------------------------------------------- # 2. Numerical — wrong number, no reliable unit signal → numerical_error # (degraded; was calculation_error in the old code) # --------------------------------------------------------------------------- def test_wrong_number_degrades_to_numerical_error() -> None: assert ( classify_error( question_type="numerical", student_answer="The answer is 350 m/s", correct_answer="340 m/s", ) == "numerical_error" ) def test_wrong_number_with_units_degrades_to_numerical_error() -> None: # Both numbers and units are present but don't match. Old code would # try to branch on calculation_error vs unit_error using fragile substring # checks. The merged bucket is safer. assert ( classify_error( question_type="numerical", student_answer="350 m/s", correct_answer="340 m/s", ) == "numerical_error" ) # --------------------------------------------------------------------------- # 3. Short text / board-style → incomplete_explanation # --------------------------------------------------------------------------- def test_short_text_with_no_keywords_is_incomplete() -> None: assert ( classify_error( question_type="short", student_answer="sound is fast", correct_answer="Equal natural frequencies cause vibration with maximum amplitude", expected_keywords=["natural frequency", "maximum amplitude"], ) == "concept_misunderstanding" ) def test_very_short_answer_is_incomplete_explanation() -> None: assert ( classify_error( question_type="short", student_answer="yes", correct_answer="A detailed explanation of resonance", expected_keywords=[], ) == "incomplete_explanation" ) # --------------------------------------------------------------------------- # 4. Correct answer — classify_error is only called on WRONG answers, so # there is no "correct" category. But verify a near-correct answer that # would be graded wrong by a rubric still classifies sensibly. # --------------------------------------------------------------------------- def test_near_correct_short_answer_is_missing_keyword() -> None: assert ( classify_error( question_type="short", student_answer="It vibrates with maximum amplitude", correct_answer="Equal natural frequencies cause vibration with maximum amplitude", expected_keywords=["natural frequency", "maximum amplitude"], ) == "missing_exam_keyword" ) # --------------------------------------------------------------------------- # 5. Formula selection — no numbers in a numerical answer # --------------------------------------------------------------------------- def test_no_numbers_in_numerical_is_formula_selection() -> None: assert ( classify_error( question_type="numerical", student_answer="I am not sure how to start", correct_answer="340 m/s", ) == "formula_selection" ) def test_no_numbers_short_text_in_numerical_is_formula_selection() -> None: assert ( classify_error( question_type="numerical", student_answer="I don't know the formula", correct_answer="340 m/s", ) == "formula_selection" ) # --------------------------------------------------------------------------- # 6. Unit error — high-confidence deterministic cases only # --------------------------------------------------------------------------- def test_exact_magnitude_missing_unit_is_unit_error() -> None: # Student wrote the right number but no unit → reliably a unit error. assert ( classify_error( question_type="numerical", student_answer="340", correct_answer="340 m/s", ) == "unit_error" ) def test_power_of_ten_ratio_is_unit_error() -> None: # 34000 vs 340 → 100x → unit conversion slip. Deterministic. assert ( classify_error( question_type="numerical", student_answer="The speed is 34000", correct_answer="340 m/s", ) == "unit_error" ) # --------------------------------------------------------------------------- # Graceful degradation: verify the old fragile cases now merge # --------------------------------------------------------------------------- def test_old_substring_unit_false_positive_merges_to_numerical() -> None: # "my answer is 5" contains "m" (metres) under the old substring check. # The new word-boundary detector must NOT fire here. assert ( classify_error( question_type="numerical", student_answer="my answer is 5", correct_answer="340 m/s", ) == "numerical_error" ) def test_mcq_blank_is_not_attempted() -> None: # Even MCQ blanks must be not_attempted, not concept_misunderstanding. assert ( classify_error( question_type="mcq", student_answer="", correct_answer="B", ) == "not_attempted" ) def test_whitespace_mcq_is_not_attempted() -> None: assert ( classify_error( question_type="mcq", student_answer=" ", correct_answer="B", ) == "not_attempted" )