document_exam_trainer / engine /validation.py
AngeloUNIMI's picture
Document Exam Trainer v5.0.0: Docker edition and local accounts
4a4df15 verified
Raw History Blame Contribute Delete
11.8 kB
"""Output guards: syntax repair is not semantic proof. Reject incomplete/ungrounded data."""
from __future__ import annotations
import json
import re
from difflib import SequenceMatcher
from .schemas import Assessment, Check, Chunk, Concept, EvaluationResult, Rubric
class OutputError(ValueError):
pass
def normalized(text: str) -> str:
return " ".join(text.split()).casefold()
def json_object(text: str) -> dict:
# Only a balanced, complete object is eligible for punctuation repair.
# Never let repair manufacture the missing end of a truncated rubric.
start = text.find("{")
if start < 0:
raise OutputError("The model did not return a structured response. Please retry.")
depth, string, escape, end = 0, False, False, None
for i in range(start, len(text)):
ch = text[i]
if string:
if escape:
escape = False
elif ch == "\\":
escape = True
elif ch == '"':
string = False
elif ch == '"':
string = True
elif ch == "{":
depth += 1
elif ch == "}":
depth -= 1
if depth == 0:
end = i + 1
break
if end is None:
raise OutputError("The model response was incomplete. Narrow the topic or retry; no incomplete rubric was accepted.")
candidate = text[start:end]
try:
data = json.loads(candidate)
except json.JSONDecodeError:
try:
import json_repair
data = json_repair.loads(candidate)
except Exception as exc:
raise OutputError("The model response could not be read safely. Please retry.") from None
if not isinstance(data, dict):
raise OutputError("The model returned the wrong response type.")
return data
def evidence_is_valid(evidence, by_id: dict[str, Chunk], role: str) -> bool:
source = by_id.get(evidence.source_id)
return bool(source and source.role == role and len(normalized(evidence.quote)) >= 8
and normalized(evidence.quote) in normalized(source.text))
def _content_tokens(text: str) -> set[str]:
# Lightweight lexical guard used only to align a model paraphrase back to
# verbatim source text. This never creates new content.
stop = {"the", "a", "an", "and", "or", "of", "to", "in", "on", "for",
"with", "is", "are", "was", "were", "be", "by", "that", "this",
"it", "as", "from", "at", "can", "may", "their", "its"}
return {t for t in re.findall(r"[a-z0-9]+", normalized(text)) if len(t) > 2 and t not in stop}
def _source_candidates(text: str, max_len: int = 650) -> list[str]:
# Prefer sentence/line-sized excerpts, plus adjacent pairs for definitions
# that cross a sentence boundary. Returned strings are exact substrings.
pieces = [m.group(0).strip() for m in re.finditer(r"[^\n.!?]+(?:[.!?]+|$)", text) if m.group(0).strip()]
out: list[str] = []
for i, piece in enumerate(pieces):
if 8 <= len(piece) <= max_len:
out.append(piece)
if i + 1 < len(pieces):
pair = (piece + " " + pieces[i + 1]).strip()
if 8 <= len(pair) <= max_len:
out.append(pair)
# Some slide/PDF extraction has no punctuation. Include bounded line blocks.
for line in (x.strip() for x in text.splitlines()):
if 8 <= len(line) <= max_len:
out.append(line)
return list(dict.fromkeys(out))
def _repair_evidence_quote(concept: Concept, evidence, source: Chunk) -> bool:
"""Align a paraphrased citation to verbatim text from the same source.
The model is allowed to choose the source passage, but the application owns
the final quotation. We only repair when lexical evidence is strong enough;
otherwise validation still fails rather than manufacturing support.
"""
target = " ".join([evidence.quote, concept.name])
target_norm = normalized(target)
target_tokens = _content_tokens(target)
quote_tokens = _content_tokens(evidence.quote)
if not target_tokens:
return False
best = None
best_score = 0.0
best_shared = 0
for candidate in _source_candidates(source.text):
cand_tokens = _content_tokens(candidate)
shared = len(target_tokens & cand_tokens)
quote_shared = len(quote_tokens & cand_tokens)
if shared < 2 or quote_shared < 2:
continue
lexical = shared / max(1, min(len(target_tokens), len(cand_tokens)))
seq = SequenceMatcher(None, target_norm, normalized(candidate)).ratio()
score = 0.65 * lexical + 0.35 * seq
if score > best_score:
best, best_score, best_shared = candidate, score, shared
if best is None or best_shared < 2 or best_score < 0.34:
return False
# Evidence schema allows 650 chars. Candidate is an exact source substring.
evidence.quote = best[:650].strip()
return True
def validate_draft(raw: dict, sources: list[Chunk]) -> Rubric:
try:
rubric = Rubric.model_validate(raw)
except Exception as exc:
raise OutputError("The generated question/rubric has invalid fields. Please retry with a narrower topic.") from None
source_map = {c.id: c for c in sources}
ids = [c.id for c in rubric.concepts]
if len(ids) != len(set(ids)):
raise OutputError("The rubric contained duplicate concept IDs. Please retry.")
for concept in rubric.concepts:
if concept.aspect not in rubric.asked_aspects:
raise OutputError("A rubric item did not match a requested aspect. Please retry.")
for evidence in concept.evidence:
if evidence_is_valid(evidence, source_map, "primary"):
continue
source = source_map.get(evidence.source_id)
if not source or source.role != "primary":
raise OutputError("A rubric citation was not supported by the supplied primary text. Please retry.")
before = evidence.quote
if not _repair_evidence_quote(concept, evidence, source) or not evidence_is_valid(evidence, source_map, "primary"):
raise OutputError("A rubric citation was not supported by the supplied primary text. Please retry.")
print(f"[Citation guard] aligned paraphrased evidence for concept={concept.id!r} source={source.id!r}", flush=True)
return rubric
def apply_audit(draft: Rubric, audit: dict) -> Rubric:
"""Apply a compact exception-only rubric audit.
v4.3 uses an exception protocol: the audit reports only IDs to drop and
severity reductions. The draft has already passed schema, aspect, and
primary-source grounding checks. Legacy decisions output is accepted too.
"""
if audit.get("question_supported") is not True:
raise OutputError("The question did not pass the source audit. Try a more specific topic.")
by_id = {c.id: c for c in draft.concepts}
expected = set(by_id)
rank = {"essential": 3, "important": 2, "minor": 1}
if "drop_ids" in audit or "lower_importance" in audit:
drop_ids = audit.get("drop_ids", [])
lower = audit.get("lower_importance", {})
if not isinstance(drop_ids, list) or not isinstance(lower, dict):
raise OutputError("The rubric audit returned invalid exception fields. Please retry.")
unknown = {x for x in drop_ids if not isinstance(x, str) or x not in expected}
unknown |= {x for x in lower if not isinstance(x, str) or x not in expected}
if unknown:
raise OutputError("The rubric audit referenced an unknown concept ID. Please retry.")
drop = set(drop_ids)
selected, names = [], set()
for cid, original in by_id.items():
if cid in drop:
continue
c = original.model_copy(deep=True)
if cid in lower:
importance = lower[cid]
if importance not in rank:
raise OutputError("Invalid audit importance category.")
if rank[importance] <= rank[c.importance]:
c.importance = importance
key = normalized(c.description)
if key not in names:
selected.append(c)
names.add(key)
if not selected:
raise OutputError("No grounded rubric concepts survived the audit. Try a different topic.")
out = draft.model_copy(deep=True)
out.concepts = selected
return out
decisions = audit.get("decisions")
if not isinstance(decisions, list):
raise OutputError("The rubric audit was incomplete. Please retry.")
valid = {}
for d in decisions:
if not isinstance(d, dict):
continue
cid = d.get("id")
if cid in expected and cid not in valid:
valid[cid] = d
missing = expected - set(valid)
if missing:
print(f"[Rubric audit] legacy audit omitted IDs={sorted(missing)}; preserving grounded draft items", flush=True)
selected, names = [], set()
for cid, original in by_id.items():
d = valid.get(cid)
if d is not None and type(d.get("keep")) is bool and not d["keep"]:
continue
c = original.model_copy(deep=True)
if d is not None:
importance = d.get("importance", c.importance)
if importance in rank and rank[importance] <= rank[c.importance]:
c.importance = importance
key = normalized(c.description)
if key not in names:
selected.append(c)
names.add(key)
if not selected:
raise OutputError("No grounded rubric concepts survived the audit. Try a different topic.")
out = draft.model_copy(deep=True)
out.concepts = selected
return out
def validate_assessment(raw: dict, rubric: Rubric, answer: str) -> EvaluationResult:
try:
result = Assessment.model_validate(raw)
except Exception as exc:
raise OutputError("The answer check was incomplete or malformed. No omissions have been inferred from missing model output; please retry.") from None
expected = [c.id for c in rubric.concepts]
actual = [c.concept_id for c in result.checks]
if len(actual) != len(set(actual)) or set(actual) != set(expected):
raise OutputError("Not every rubric item was checked reliably. Please retry. Missing model checks are NOT treated as student omissions.")
checks = {c.concept_id: c for c in result.checks}
ordered, warnings, seen_gaps = [], [], set()
for cid in expected:
check = checks[cid]
if check.status in ("covered", "partial", "incorrect"):
if not check.answer_evidence or normalized(check.answer_evidence) not in normalized(answer):
check.status = "uncertain"
warnings.append("One content check lacked verifiable evidence from the answer and was not turned into an omission.")
if check.status in ("partial", "missing", "incorrect"):
if not check.missing_detail or not check.what_to_add:
check.status = "uncertain"
warnings.append("One content check was too incomplete to report reliably.")
else:
key = normalized(check.missing_detail)
if key in seen_gaps:
check.status = "uncertain"
warnings.append("A duplicate description of the same gap was suppressed.")
seen_gaps.add(key)
if check.status == "uncertain" and not warnings:
warnings.append("The model was uncertain about part of this answer. Uncertain checks are not omissions.")
ordered.append(check)
return EvaluationResult(ordered, list(dict.fromkeys(warnings)))