Spaces:
Running on Zero
Running on Zero
Download engine/validation.py from AngeloUNIMI/document_exam_trainer: direct link, hf CLI and curl.
- Browser
- Download file 11.8 kB
-
https://huggingface.co/spaces/AngeloUNIMI/document_exam_trainer/resolve/main/engine/validation.py
- Command line
-
hf download hf://spaces/AngeloUNIMI/document_exam_trainer/engine/validation.py
-
curl -L -o validation.py https://huggingface.co/spaces/AngeloUNIMI/document_exam_trainer/resolve/main/engine/validation.py
11.8 kB
| """Output guards: syntax repair is not semantic proof. Reject incomplete/ungrounded data.""" | |
| from __future__ import annotations | |
| import json | |
| import re | |
| from difflib import SequenceMatcher | |
| from .schemas import Assessment, Check, Chunk, Concept, EvaluationResult, Rubric | |
| class OutputError(ValueError): | |
| pass | |
| def normalized(text: str) -> str: | |
| return " ".join(text.split()).casefold() | |
| def json_object(text: str) -> dict: | |
| # Only a balanced, complete object is eligible for punctuation repair. | |
| # Never let repair manufacture the missing end of a truncated rubric. | |
| start = text.find("{") | |
| if start < 0: | |
| raise OutputError("The model did not return a structured response. Please retry.") | |
| depth, string, escape, end = 0, False, False, None | |
| for i in range(start, len(text)): | |
| ch = text[i] | |
| if string: | |
| if escape: | |
| escape = False | |
| elif ch == "\\": | |
| escape = True | |
| elif ch == '"': | |
| string = False | |
| elif ch == '"': | |
| string = True | |
| elif ch == "{": | |
| depth += 1 | |
| elif ch == "}": | |
| depth -= 1 | |
| if depth == 0: | |
| end = i + 1 | |
| break | |
| if end is None: | |
| raise OutputError("The model response was incomplete. Narrow the topic or retry; no incomplete rubric was accepted.") | |
| candidate = text[start:end] | |
| try: | |
| data = json.loads(candidate) | |
| except json.JSONDecodeError: | |
| try: | |
| import json_repair | |
| data = json_repair.loads(candidate) | |
| except Exception as exc: | |
| raise OutputError("The model response could not be read safely. Please retry.") from None | |
| if not isinstance(data, dict): | |
| raise OutputError("The model returned the wrong response type.") | |
| return data | |
| def evidence_is_valid(evidence, by_id: dict[str, Chunk], role: str) -> bool: | |
| source = by_id.get(evidence.source_id) | |
| return bool(source and source.role == role and len(normalized(evidence.quote)) >= 8 | |
| and normalized(evidence.quote) in normalized(source.text)) | |
| def _content_tokens(text: str) -> set[str]: | |
| # Lightweight lexical guard used only to align a model paraphrase back to | |
| # verbatim source text. This never creates new content. | |
| stop = {"the", "a", "an", "and", "or", "of", "to", "in", "on", "for", | |
| "with", "is", "are", "was", "were", "be", "by", "that", "this", | |
| "it", "as", "from", "at", "can", "may", "their", "its"} | |
| return {t for t in re.findall(r"[a-z0-9]+", normalized(text)) if len(t) > 2 and t not in stop} | |
| def _source_candidates(text: str, max_len: int = 650) -> list[str]: | |
| # Prefer sentence/line-sized excerpts, plus adjacent pairs for definitions | |
| # that cross a sentence boundary. Returned strings are exact substrings. | |
| pieces = [m.group(0).strip() for m in re.finditer(r"[^\n.!?]+(?:[.!?]+|$)", text) if m.group(0).strip()] | |
| out: list[str] = [] | |
| for i, piece in enumerate(pieces): | |
| if 8 <= len(piece) <= max_len: | |
| out.append(piece) | |
| if i + 1 < len(pieces): | |
| pair = (piece + " " + pieces[i + 1]).strip() | |
| if 8 <= len(pair) <= max_len: | |
| out.append(pair) | |
| # Some slide/PDF extraction has no punctuation. Include bounded line blocks. | |
| for line in (x.strip() for x in text.splitlines()): | |
| if 8 <= len(line) <= max_len: | |
| out.append(line) | |
| return list(dict.fromkeys(out)) | |
| def _repair_evidence_quote(concept: Concept, evidence, source: Chunk) -> bool: | |
| """Align a paraphrased citation to verbatim text from the same source. | |
| The model is allowed to choose the source passage, but the application owns | |
| the final quotation. We only repair when lexical evidence is strong enough; | |
| otherwise validation still fails rather than manufacturing support. | |
| """ | |
| target = " ".join([evidence.quote, concept.name]) | |
| target_norm = normalized(target) | |
| target_tokens = _content_tokens(target) | |
| quote_tokens = _content_tokens(evidence.quote) | |
| if not target_tokens: | |
| return False | |
| best = None | |
| best_score = 0.0 | |
| best_shared = 0 | |
| for candidate in _source_candidates(source.text): | |
| cand_tokens = _content_tokens(candidate) | |
| shared = len(target_tokens & cand_tokens) | |
| quote_shared = len(quote_tokens & cand_tokens) | |
| if shared < 2 or quote_shared < 2: | |
| continue | |
| lexical = shared / max(1, min(len(target_tokens), len(cand_tokens))) | |
| seq = SequenceMatcher(None, target_norm, normalized(candidate)).ratio() | |
| score = 0.65 * lexical + 0.35 * seq | |
| if score > best_score: | |
| best, best_score, best_shared = candidate, score, shared | |
| if best is None or best_shared < 2 or best_score < 0.34: | |
| return False | |
| # Evidence schema allows 650 chars. Candidate is an exact source substring. | |
| evidence.quote = best[:650].strip() | |
| return True | |
| def validate_draft(raw: dict, sources: list[Chunk]) -> Rubric: | |
| try: | |
| rubric = Rubric.model_validate(raw) | |
| except Exception as exc: | |
| raise OutputError("The generated question/rubric has invalid fields. Please retry with a narrower topic.") from None | |
| source_map = {c.id: c for c in sources} | |
| ids = [c.id for c in rubric.concepts] | |
| if len(ids) != len(set(ids)): | |
| raise OutputError("The rubric contained duplicate concept IDs. Please retry.") | |
| for concept in rubric.concepts: | |
| if concept.aspect not in rubric.asked_aspects: | |
| raise OutputError("A rubric item did not match a requested aspect. Please retry.") | |
| for evidence in concept.evidence: | |
| if evidence_is_valid(evidence, source_map, "primary"): | |
| continue | |
| source = source_map.get(evidence.source_id) | |
| if not source or source.role != "primary": | |
| raise OutputError("A rubric citation was not supported by the supplied primary text. Please retry.") | |
| before = evidence.quote | |
| if not _repair_evidence_quote(concept, evidence, source) or not evidence_is_valid(evidence, source_map, "primary"): | |
| raise OutputError("A rubric citation was not supported by the supplied primary text. Please retry.") | |
| print(f"[Citation guard] aligned paraphrased evidence for concept={concept.id!r} source={source.id!r}", flush=True) | |
| return rubric | |
| def apply_audit(draft: Rubric, audit: dict) -> Rubric: | |
| """Apply a compact exception-only rubric audit. | |
| v4.3 uses an exception protocol: the audit reports only IDs to drop and | |
| severity reductions. The draft has already passed schema, aspect, and | |
| primary-source grounding checks. Legacy decisions output is accepted too. | |
| """ | |
| if audit.get("question_supported") is not True: | |
| raise OutputError("The question did not pass the source audit. Try a more specific topic.") | |
| by_id = {c.id: c for c in draft.concepts} | |
| expected = set(by_id) | |
| rank = {"essential": 3, "important": 2, "minor": 1} | |
| if "drop_ids" in audit or "lower_importance" in audit: | |
| drop_ids = audit.get("drop_ids", []) | |
| lower = audit.get("lower_importance", {}) | |
| if not isinstance(drop_ids, list) or not isinstance(lower, dict): | |
| raise OutputError("The rubric audit returned invalid exception fields. Please retry.") | |
| unknown = {x for x in drop_ids if not isinstance(x, str) or x not in expected} | |
| unknown |= {x for x in lower if not isinstance(x, str) or x not in expected} | |
| if unknown: | |
| raise OutputError("The rubric audit referenced an unknown concept ID. Please retry.") | |
| drop = set(drop_ids) | |
| selected, names = [], set() | |
| for cid, original in by_id.items(): | |
| if cid in drop: | |
| continue | |
| c = original.model_copy(deep=True) | |
| if cid in lower: | |
| importance = lower[cid] | |
| if importance not in rank: | |
| raise OutputError("Invalid audit importance category.") | |
| if rank[importance] <= rank[c.importance]: | |
| c.importance = importance | |
| key = normalized(c.description) | |
| if key not in names: | |
| selected.append(c) | |
| names.add(key) | |
| if not selected: | |
| raise OutputError("No grounded rubric concepts survived the audit. Try a different topic.") | |
| out = draft.model_copy(deep=True) | |
| out.concepts = selected | |
| return out | |
| decisions = audit.get("decisions") | |
| if not isinstance(decisions, list): | |
| raise OutputError("The rubric audit was incomplete. Please retry.") | |
| valid = {} | |
| for d in decisions: | |
| if not isinstance(d, dict): | |
| continue | |
| cid = d.get("id") | |
| if cid in expected and cid not in valid: | |
| valid[cid] = d | |
| missing = expected - set(valid) | |
| if missing: | |
| print(f"[Rubric audit] legacy audit omitted IDs={sorted(missing)}; preserving grounded draft items", flush=True) | |
| selected, names = [], set() | |
| for cid, original in by_id.items(): | |
| d = valid.get(cid) | |
| if d is not None and type(d.get("keep")) is bool and not d["keep"]: | |
| continue | |
| c = original.model_copy(deep=True) | |
| if d is not None: | |
| importance = d.get("importance", c.importance) | |
| if importance in rank and rank[importance] <= rank[c.importance]: | |
| c.importance = importance | |
| key = normalized(c.description) | |
| if key not in names: | |
| selected.append(c) | |
| names.add(key) | |
| if not selected: | |
| raise OutputError("No grounded rubric concepts survived the audit. Try a different topic.") | |
| out = draft.model_copy(deep=True) | |
| out.concepts = selected | |
| return out | |
| def validate_assessment(raw: dict, rubric: Rubric, answer: str) -> EvaluationResult: | |
| try: | |
| result = Assessment.model_validate(raw) | |
| except Exception as exc: | |
| raise OutputError("The answer check was incomplete or malformed. No omissions have been inferred from missing model output; please retry.") from None | |
| expected = [c.id for c in rubric.concepts] | |
| actual = [c.concept_id for c in result.checks] | |
| if len(actual) != len(set(actual)) or set(actual) != set(expected): | |
| raise OutputError("Not every rubric item was checked reliably. Please retry. Missing model checks are NOT treated as student omissions.") | |
| checks = {c.concept_id: c for c in result.checks} | |
| ordered, warnings, seen_gaps = [], [], set() | |
| for cid in expected: | |
| check = checks[cid] | |
| if check.status in ("covered", "partial", "incorrect"): | |
| if not check.answer_evidence or normalized(check.answer_evidence) not in normalized(answer): | |
| check.status = "uncertain" | |
| warnings.append("One content check lacked verifiable evidence from the answer and was not turned into an omission.") | |
| if check.status in ("partial", "missing", "incorrect"): | |
| if not check.missing_detail or not check.what_to_add: | |
| check.status = "uncertain" | |
| warnings.append("One content check was too incomplete to report reliably.") | |
| else: | |
| key = normalized(check.missing_detail) | |
| if key in seen_gaps: | |
| check.status = "uncertain" | |
| warnings.append("A duplicate description of the same gap was suppressed.") | |
| seen_gaps.add(key) | |
| if check.status == "uncertain" and not warnings: | |
| warnings.append("The model was uncertain about part of this answer. Uncertain checks are not omissions.") | |
| ordered.append(check) | |
| return EvaluationResult(ordered, list(dict.fromkeys(warnings))) | |