DocDoeAI / tests /test_error_classification_harness.py
asnannp's picture
deploy: sync backend to Space root (learn-lesson HF cache fix)
65d68f5
Raw History Blame Contribute Delete
6.81 kB
"""Error-classification validation harness (Wave 1, Item 10).
Deterministic rules only β€” no LLM, no model calls, no network.
Tests 6 canonical error-classification cases and verifies the v1.1 graceful-
degradation rule: where deterministic rules cannot reliably separate
``unit_error`` from ``calculation_error``, the distinction merges into the
broader ``numerical_error`` bucket (Product Strategy Β§15.3).
"""
from __future__ import annotations
from app.services.mastery_engine import classify_error
# ---------------------------------------------------------------------------
# 1. Blank / whitespace β†’ not_attempted
# ---------------------------------------------------------------------------
def test_blank_answer_is_not_attempted() -> None:
assert (
classify_error(
question_type="numerical",
student_answer="",
correct_answer="340 m/s",
)
== "not_attempted"
)
def test_whitespace_only_answer_is_not_attempted() -> None:
assert (
classify_error(
question_type="numerical",
student_answer=" \t \n ",
correct_answer="340 m/s",
)
== "not_attempted"
)
# ---------------------------------------------------------------------------
# 2. Numerical β€” wrong number, no reliable unit signal β†’ numerical_error
# (degraded; was calculation_error in the old code)
# ---------------------------------------------------------------------------
def test_wrong_number_degrades_to_numerical_error() -> None:
assert (
classify_error(
question_type="numerical",
student_answer="The answer is 350 m/s",
correct_answer="340 m/s",
)
== "numerical_error"
)
def test_wrong_number_with_units_degrades_to_numerical_error() -> None:
# Both numbers and units are present but don't match. Old code would
# try to branch on calculation_error vs unit_error using fragile substring
# checks. The merged bucket is safer.
assert (
classify_error(
question_type="numerical",
student_answer="350 m/s",
correct_answer="340 m/s",
)
== "numerical_error"
)
# ---------------------------------------------------------------------------
# 3. Short text / board-style β†’ incomplete_explanation
# ---------------------------------------------------------------------------
def test_short_text_with_no_keywords_is_incomplete() -> None:
assert (
classify_error(
question_type="short",
student_answer="sound is fast",
correct_answer="Equal natural frequencies cause vibration with maximum amplitude",
expected_keywords=["natural frequency", "maximum amplitude"],
)
== "concept_misunderstanding"
)
def test_very_short_answer_is_incomplete_explanation() -> None:
assert (
classify_error(
question_type="short",
student_answer="yes",
correct_answer="A detailed explanation of resonance",
expected_keywords=[],
)
== "incomplete_explanation"
)
# ---------------------------------------------------------------------------
# 4. Correct answer β€” classify_error is only called on WRONG answers, so
# there is no "correct" category. But verify a near-correct answer that
# would be graded wrong by a rubric still classifies sensibly.
# ---------------------------------------------------------------------------
def test_near_correct_short_answer_is_missing_keyword() -> None:
assert (
classify_error(
question_type="short",
student_answer="It vibrates with maximum amplitude",
correct_answer="Equal natural frequencies cause vibration with maximum amplitude",
expected_keywords=["natural frequency", "maximum amplitude"],
)
== "missing_exam_keyword"
)
# ---------------------------------------------------------------------------
# 5. Formula selection β€” no numbers in a numerical answer
# ---------------------------------------------------------------------------
def test_no_numbers_in_numerical_is_formula_selection() -> None:
assert (
classify_error(
question_type="numerical",
student_answer="I am not sure how to start",
correct_answer="340 m/s",
)
== "formula_selection"
)
def test_no_numbers_short_text_in_numerical_is_formula_selection() -> None:
assert (
classify_error(
question_type="numerical",
student_answer="I don't know the formula",
correct_answer="340 m/s",
)
== "formula_selection"
)
# ---------------------------------------------------------------------------
# 6. Unit error β€” high-confidence deterministic cases only
# ---------------------------------------------------------------------------
def test_exact_magnitude_missing_unit_is_unit_error() -> None:
# Student wrote the right number but no unit β†’ reliably a unit error.
assert (
classify_error(
question_type="numerical",
student_answer="340",
correct_answer="340 m/s",
)
== "unit_error"
)
def test_power_of_ten_ratio_is_unit_error() -> None:
# 34000 vs 340 β†’ 100x β†’ unit conversion slip. Deterministic.
assert (
classify_error(
question_type="numerical",
student_answer="The speed is 34000",
correct_answer="340 m/s",
)
== "unit_error"
)
# ---------------------------------------------------------------------------
# Graceful degradation: verify the old fragile cases now merge
# ---------------------------------------------------------------------------
def test_old_substring_unit_false_positive_merges_to_numerical() -> None:
# "my answer is 5" contains "m" (metres) under the old substring check.
# The new word-boundary detector must NOT fire here.
assert (
classify_error(
question_type="numerical",
student_answer="my answer is 5",
correct_answer="340 m/s",
)
== "numerical_error"
)
def test_mcq_blank_is_not_attempted() -> None:
# Even MCQ blanks must be not_attempted, not concept_misunderstanding.
assert (
classify_error(
question_type="mcq",
student_answer="",
correct_answer="B",
)
== "not_attempted"
)
def test_whitespace_mcq_is_not_attempted() -> None:
assert (
classify_error(
question_type="mcq",
student_answer=" ",
correct_answer="B",
)
== "not_attempted"
)