Download tests/test_error_classification_harness.py from asnannp/DocDoeAI: direct link, hf CLI and curl.
- Browser
- Download file 6.81 kB
-
https://huggingface.co/spaces/asnannp/DocDoeAI/resolve/main/tests/test_error_classification_harness.py
- Command line
-
hf download hf://spaces/asnannp/DocDoeAI/tests/test_error_classification_harness.py
-
curl -L -o test_error_classification_harness.py https://huggingface.co/spaces/asnannp/DocDoeAI/resolve/main/tests/test_error_classification_harness.py
6.81 kB
| """Error-classification validation harness (Wave 1, Item 10). | |
| Deterministic rules only β no LLM, no model calls, no network. | |
| Tests 6 canonical error-classification cases and verifies the v1.1 graceful- | |
| degradation rule: where deterministic rules cannot reliably separate | |
| ``unit_error`` from ``calculation_error``, the distinction merges into the | |
| broader ``numerical_error`` bucket (Product Strategy Β§15.3). | |
| """ | |
| from __future__ import annotations | |
| from app.services.mastery_engine import classify_error | |
| # --------------------------------------------------------------------------- | |
| # 1. Blank / whitespace β not_attempted | |
| # --------------------------------------------------------------------------- | |
| def test_blank_answer_is_not_attempted() -> None: | |
| assert ( | |
| classify_error( | |
| question_type="numerical", | |
| student_answer="", | |
| correct_answer="340 m/s", | |
| ) | |
| == "not_attempted" | |
| ) | |
| def test_whitespace_only_answer_is_not_attempted() -> None: | |
| assert ( | |
| classify_error( | |
| question_type="numerical", | |
| student_answer=" \t \n ", | |
| correct_answer="340 m/s", | |
| ) | |
| == "not_attempted" | |
| ) | |
| # --------------------------------------------------------------------------- | |
| # 2. Numerical β wrong number, no reliable unit signal β numerical_error | |
| # (degraded; was calculation_error in the old code) | |
| # --------------------------------------------------------------------------- | |
| def test_wrong_number_degrades_to_numerical_error() -> None: | |
| assert ( | |
| classify_error( | |
| question_type="numerical", | |
| student_answer="The answer is 350 m/s", | |
| correct_answer="340 m/s", | |
| ) | |
| == "numerical_error" | |
| ) | |
| def test_wrong_number_with_units_degrades_to_numerical_error() -> None: | |
| # Both numbers and units are present but don't match. Old code would | |
| # try to branch on calculation_error vs unit_error using fragile substring | |
| # checks. The merged bucket is safer. | |
| assert ( | |
| classify_error( | |
| question_type="numerical", | |
| student_answer="350 m/s", | |
| correct_answer="340 m/s", | |
| ) | |
| == "numerical_error" | |
| ) | |
| # --------------------------------------------------------------------------- | |
| # 3. Short text / board-style β incomplete_explanation | |
| # --------------------------------------------------------------------------- | |
| def test_short_text_with_no_keywords_is_incomplete() -> None: | |
| assert ( | |
| classify_error( | |
| question_type="short", | |
| student_answer="sound is fast", | |
| correct_answer="Equal natural frequencies cause vibration with maximum amplitude", | |
| expected_keywords=["natural frequency", "maximum amplitude"], | |
| ) | |
| == "concept_misunderstanding" | |
| ) | |
| def test_very_short_answer_is_incomplete_explanation() -> None: | |
| assert ( | |
| classify_error( | |
| question_type="short", | |
| student_answer="yes", | |
| correct_answer="A detailed explanation of resonance", | |
| expected_keywords=[], | |
| ) | |
| == "incomplete_explanation" | |
| ) | |
| # --------------------------------------------------------------------------- | |
| # 4. Correct answer β classify_error is only called on WRONG answers, so | |
| # there is no "correct" category. But verify a near-correct answer that | |
| # would be graded wrong by a rubric still classifies sensibly. | |
| # --------------------------------------------------------------------------- | |
| def test_near_correct_short_answer_is_missing_keyword() -> None: | |
| assert ( | |
| classify_error( | |
| question_type="short", | |
| student_answer="It vibrates with maximum amplitude", | |
| correct_answer="Equal natural frequencies cause vibration with maximum amplitude", | |
| expected_keywords=["natural frequency", "maximum amplitude"], | |
| ) | |
| == "missing_exam_keyword" | |
| ) | |
| # --------------------------------------------------------------------------- | |
| # 5. Formula selection β no numbers in a numerical answer | |
| # --------------------------------------------------------------------------- | |
| def test_no_numbers_in_numerical_is_formula_selection() -> None: | |
| assert ( | |
| classify_error( | |
| question_type="numerical", | |
| student_answer="I am not sure how to start", | |
| correct_answer="340 m/s", | |
| ) | |
| == "formula_selection" | |
| ) | |
| def test_no_numbers_short_text_in_numerical_is_formula_selection() -> None: | |
| assert ( | |
| classify_error( | |
| question_type="numerical", | |
| student_answer="I don't know the formula", | |
| correct_answer="340 m/s", | |
| ) | |
| == "formula_selection" | |
| ) | |
| # --------------------------------------------------------------------------- | |
| # 6. Unit error β high-confidence deterministic cases only | |
| # --------------------------------------------------------------------------- | |
| def test_exact_magnitude_missing_unit_is_unit_error() -> None: | |
| # Student wrote the right number but no unit β reliably a unit error. | |
| assert ( | |
| classify_error( | |
| question_type="numerical", | |
| student_answer="340", | |
| correct_answer="340 m/s", | |
| ) | |
| == "unit_error" | |
| ) | |
| def test_power_of_ten_ratio_is_unit_error() -> None: | |
| # 34000 vs 340 β 100x β unit conversion slip. Deterministic. | |
| assert ( | |
| classify_error( | |
| question_type="numerical", | |
| student_answer="The speed is 34000", | |
| correct_answer="340 m/s", | |
| ) | |
| == "unit_error" | |
| ) | |
| # --------------------------------------------------------------------------- | |
| # Graceful degradation: verify the old fragile cases now merge | |
| # --------------------------------------------------------------------------- | |
| def test_old_substring_unit_false_positive_merges_to_numerical() -> None: | |
| # "my answer is 5" contains "m" (metres) under the old substring check. | |
| # The new word-boundary detector must NOT fire here. | |
| assert ( | |
| classify_error( | |
| question_type="numerical", | |
| student_answer="my answer is 5", | |
| correct_answer="340 m/s", | |
| ) | |
| == "numerical_error" | |
| ) | |
| def test_mcq_blank_is_not_attempted() -> None: | |
| # Even MCQ blanks must be not_attempted, not concept_misunderstanding. | |
| assert ( | |
| classify_error( | |
| question_type="mcq", | |
| student_answer="", | |
| correct_answer="B", | |
| ) | |
| == "not_attempted" | |
| ) | |
| def test_whitespace_mcq_is_not_attempted() -> None: | |
| assert ( | |
| classify_error( | |
| question_type="mcq", | |
| student_answer=" ", | |
| correct_answer="B", | |
| ) | |
| == "not_attempted" | |
| ) | |