Spaces:
Running
Running
github-actions[bot] commited on
Commit ·
2b2e5e8
1
Parent(s): 70c6e17
deploy: backend from 6d07014
Browse files- rag/curriculum_rag.py +24 -10
- routes/fun_modules_routes.py +1 -30
- routes/jev_routes.py +4 -2
- services/jev_client.py +69 -14
- tests/test_fun_modules_routes.py +41 -17
rag/curriculum_rag.py
CHANGED
|
@@ -7,6 +7,13 @@ from __future__ import annotations
|
|
| 7 |
import re
|
| 8 |
from typing import Any, Dict, List, Optional, Tuple
|
| 9 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 10 |
|
| 11 |
def _normalize_subject(subject: Optional[str]) -> Optional[str]:
|
| 12 |
if not subject:
|
|
@@ -192,7 +199,12 @@ def _metadata_matches(
|
|
| 192 |
return True
|
| 193 |
|
| 194 |
|
| 195 |
-
def _row_from_chunk(
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 196 |
ret_storage = str(md.get("storage_path") or "").strip()
|
| 197 |
ret_source_file = str(md.get("source_file") or "").strip()
|
| 198 |
ret_source_path = str(md.get("source_path") or "").strip()
|
|
@@ -293,7 +305,7 @@ def _retrieve_exact_file_chunks(
|
|
| 293 |
lesson_id: str | None,
|
| 294 |
competency_code: str | None,
|
| 295 |
top_k: int,
|
| 296 |
-
) -> List[
|
| 297 |
"""Exact-match retrieval for one source file via collection.get (server-side
|
| 298 |
metadata filtering works; only id-resolving vector reads are broken)."""
|
| 299 |
cand_paths, cand_files = _normalize_storage_candidates(storage_path)
|
|
@@ -325,7 +337,7 @@ def _retrieve_exact_file_chunks(
|
|
| 325 |
):
|
| 326 |
continue
|
| 327 |
kept.append((idx, md))
|
| 328 |
-
rows: List[
|
| 329 |
if kept:
|
| 330 |
texts = [str(documents[idx]) if idx < len(documents) else "" for idx, _ in kept]
|
| 331 |
try:
|
|
@@ -355,7 +367,7 @@ def retrieve_curriculum_context(
|
|
| 355 |
top_k: int = 8,
|
| 356 |
grade_level: str | None = None,
|
| 357 |
**kwargs: Any,
|
| 358 |
-
) -> list[
|
| 359 |
from rag.vectorstore_loader import get_vectorstore_components
|
| 360 |
|
| 361 |
_, collection, embedder = get_vectorstore_components()
|
|
@@ -395,7 +407,7 @@ def retrieve_curriculum_context(
|
|
| 395 |
metadatas = (result.get("metadatas") or [[]])[0]
|
| 396 |
distances = (result.get("distances") or [[]])[0]
|
| 397 |
|
| 398 |
-
rows: List[
|
| 399 |
for idx, content in enumerate(documents):
|
| 400 |
md = metadatas[idx] if idx < len(metadatas) and isinstance(metadatas[idx], dict) else {}
|
| 401 |
distance = float(distances[idx]) if idx < len(distances) else 1.0
|
|
@@ -463,7 +475,7 @@ def retrieve_lesson_pdf_context(
|
|
| 463 |
competency_code: str | None = None,
|
| 464 |
storage_path: str | None = None,
|
| 465 |
top_k: int = 8,
|
| 466 |
-
) -> Tuple[list[
|
| 467 |
"""Retrieve chunks by storage_path exact match + semantic ranking; fallback to general query.
|
| 468 |
|
| 469 |
NOTE: Curriculum PDF chunks are often tagged with quarter=1 or 0 even when covering other topics.
|
|
@@ -491,7 +503,7 @@ def retrieve_lesson_pdf_context(
|
|
| 491 |
else:
|
| 492 |
search_query = stripped_topic
|
| 493 |
|
| 494 |
-
exact_chunks: list[
|
| 495 |
if storage_path:
|
| 496 |
# Try 1: Exact match with storage_path + quarter
|
| 497 |
if quarter and quarter > 0:
|
|
@@ -693,8 +705,10 @@ def build_problem_generation_prompt(topic: str, difficulty: str, curriculum_chun
|
|
| 693 |
)
|
| 694 |
|
| 695 |
|
| 696 |
-
def build_analysis_curriculum_context(
|
| 697 |
-
|
|
|
|
|
|
|
| 698 |
for weak_topic in weak_topics:
|
| 699 |
rows = retrieve_curriculum_context(
|
| 700 |
query=f"DepEd learning competency for {weak_topic}",
|
|
@@ -706,4 +720,4 @@ def build_analysis_curriculum_context(weak_topics: list[str], subject: str) -> l
|
|
| 706 |
key = f"{row.get('source_file')}::{row.get('page')}::{row.get('content', '')[:80]}"
|
| 707 |
if key not in dedup:
|
| 708 |
dedup[key] = row
|
| 709 |
-
return list(dedup.values())
|
|
|
|
| 7 |
import re
|
| 8 |
from typing import Any, Dict, List, Optional, Tuple
|
| 9 |
|
| 10 |
+
# Canonical retrieval-row keys: content, subject, quarter, content_domain,
|
| 11 |
+
# chunk_type, source_file, storage_path, module_id, lesson_id,
|
| 12 |
+
# competency_code, page, score. Plain-dict alias (not TypedDict) so existing
|
| 13 |
+
# List[Dict[str, Any]] consumers keep typechecking (list invariance).
|
| 14 |
+
# ponytail: upgrade to TypedDict when callers move off Dict[str, Any].
|
| 15 |
+
CurriculumChunk = Dict[str, Any]
|
| 16 |
+
|
| 17 |
|
| 18 |
def _normalize_subject(subject: Optional[str]) -> Optional[str]:
|
| 19 |
if not subject:
|
|
|
|
| 199 |
return True
|
| 200 |
|
| 201 |
|
| 202 |
+
def _row_from_chunk(
|
| 203 |
+
content: Any,
|
| 204 |
+
md: Dict[str, Any],
|
| 205 |
+
distance: float,
|
| 206 |
+
storage_path: Optional[str] = None,
|
| 207 |
+
) -> CurriculumChunk:
|
| 208 |
ret_storage = str(md.get("storage_path") or "").strip()
|
| 209 |
ret_source_file = str(md.get("source_file") or "").strip()
|
| 210 |
ret_source_path = str(md.get("source_path") or "").strip()
|
|
|
|
| 305 |
lesson_id: str | None,
|
| 306 |
competency_code: str | None,
|
| 307 |
top_k: int,
|
| 308 |
+
) -> List[CurriculumChunk]:
|
| 309 |
"""Exact-match retrieval for one source file via collection.get (server-side
|
| 310 |
metadata filtering works; only id-resolving vector reads are broken)."""
|
| 311 |
cand_paths, cand_files = _normalize_storage_candidates(storage_path)
|
|
|
|
| 337 |
):
|
| 338 |
continue
|
| 339 |
kept.append((idx, md))
|
| 340 |
+
rows: List[CurriculumChunk] = []
|
| 341 |
if kept:
|
| 342 |
texts = [str(documents[idx]) if idx < len(documents) else "" for idx, _ in kept]
|
| 343 |
try:
|
|
|
|
| 367 |
top_k: int = 8,
|
| 368 |
grade_level: str | None = None,
|
| 369 |
**kwargs: Any,
|
| 370 |
+
) -> list[CurriculumChunk]:
|
| 371 |
from rag.vectorstore_loader import get_vectorstore_components
|
| 372 |
|
| 373 |
_, collection, embedder = get_vectorstore_components()
|
|
|
|
| 407 |
metadatas = (result.get("metadatas") or [[]])[0]
|
| 408 |
distances = (result.get("distances") or [[]])[0]
|
| 409 |
|
| 410 |
+
rows: List[CurriculumChunk] = []
|
| 411 |
for idx, content in enumerate(documents):
|
| 412 |
md = metadatas[idx] if idx < len(metadatas) and isinstance(metadatas[idx], dict) else {}
|
| 413 |
distance = float(distances[idx]) if idx < len(distances) else 1.0
|
|
|
|
| 475 |
competency_code: str | None = None,
|
| 476 |
storage_path: str | None = None,
|
| 477 |
top_k: int = 8,
|
| 478 |
+
) -> Tuple[list[CurriculumChunk], str]:
|
| 479 |
"""Retrieve chunks by storage_path exact match + semantic ranking; fallback to general query.
|
| 480 |
|
| 481 |
NOTE: Curriculum PDF chunks are often tagged with quarter=1 or 0 even when covering other topics.
|
|
|
|
| 503 |
else:
|
| 504 |
search_query = stripped_topic
|
| 505 |
|
| 506 |
+
exact_chunks: list[CurriculumChunk] = []
|
| 507 |
if storage_path:
|
| 508 |
# Try 1: Exact match with storage_path + quarter
|
| 509 |
if quarter and quarter > 0:
|
|
|
|
| 705 |
)
|
| 706 |
|
| 707 |
|
| 708 |
+
def build_analysis_curriculum_context(
|
| 709 |
+
weak_topics: list[str], subject: str
|
| 710 |
+
) -> list[CurriculumChunk]:
|
| 711 |
+
dedup: Dict[str, CurriculumChunk] = {}
|
| 712 |
for weak_topic in weak_topics:
|
| 713 |
rows = retrieve_curriculum_context(
|
| 714 |
query=f"DepEd learning competency for {weak_topic}",
|
|
|
|
| 720 |
key = f"{row.get('source_file')}::{row.get('page')}::{row.get('content', '')[:80]}"
|
| 721 |
if key not in dedup:
|
| 722 |
dedup[key] = row
|
| 723 |
+
return list(dedup.values())
|
routes/fun_modules_routes.py
CHANGED
|
@@ -1,6 +1,5 @@
|
|
| 1 |
-
from __future__ import annotations
|
| 2 |
|
| 3 |
-
import os
|
| 4 |
import re
|
| 5 |
from typing import Any
|
| 6 |
|
|
@@ -33,12 +32,6 @@ class MasteryRecordRequest(BaseModel):
|
|
| 33 |
score: float = Field(ge=0, le=1)
|
| 34 |
|
| 35 |
|
| 36 |
-
class JevVerifyRequest(BaseModel):
|
| 37 |
-
referenceText: str = Field(min_length=1)
|
| 38 |
-
generatedText: str = Field(min_length=1)
|
| 39 |
-
claimType: str = Field(min_length=1)
|
| 40 |
-
|
| 41 |
-
|
| 42 |
class GenerateModuleResponse(BaseModel):
|
| 43 |
moduleId: str
|
| 44 |
title: str
|
|
@@ -52,13 +45,6 @@ class MasteryRecordResponse(BaseModel):
|
|
| 52 |
xpAwarded: int
|
| 53 |
|
| 54 |
|
| 55 |
-
class JevVerifyResponse(BaseModel):
|
| 56 |
-
verified: bool
|
| 57 |
-
pCorrect: float
|
| 58 |
-
pLeak: float
|
| 59 |
-
action: str
|
| 60 |
-
|
| 61 |
-
|
| 62 |
def _api_error(status_code: int, error: str, error_type: str) -> HTTPException:
|
| 63 |
return HTTPException(
|
| 64 |
status_code=status_code,
|
|
@@ -144,18 +130,3 @@ async def record_mastery(payload: MasteryRecordRequest) -> dict[str, Any]:
|
|
| 144 |
"unlockedModules": [],
|
| 145 |
"xpAwarded": 30 if payload.correct else 0,
|
| 146 |
}
|
| 147 |
-
|
| 148 |
-
|
| 149 |
-
@router.post("/jev/verify", response_model=JevVerifyResponse)
|
| 150 |
-
async def verify_with_jev(payload: JevVerifyRequest) -> dict[str, Any]:
|
| 151 |
-
os.getenv("TYPESAFE_API_KEY")
|
| 152 |
-
reference = " ".join(payload.referenceText.lower().split())
|
| 153 |
-
generated = " ".join(payload.generatedText.lower().split())
|
| 154 |
-
verified = reference == generated or reference in generated
|
| 155 |
-
leak = "answer" in generated and payload.claimType.lower() in {"hint", "practice"}
|
| 156 |
-
return {
|
| 157 |
-
"verified": verified,
|
| 158 |
-
"pCorrect": 0.95 if verified else 0.35,
|
| 159 |
-
"pLeak": 0.9 if leak else 0.05,
|
| 160 |
-
"action": "allow_with_fallback",
|
| 161 |
-
}
|
|
|
|
| 1 |
+
from __future__ import annotations
|
| 2 |
|
|
|
|
| 3 |
import re
|
| 4 |
from typing import Any
|
| 5 |
|
|
|
|
| 32 |
score: float = Field(ge=0, le=1)
|
| 33 |
|
| 34 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 35 |
class GenerateModuleResponse(BaseModel):
|
| 36 |
moduleId: str
|
| 37 |
title: str
|
|
|
|
| 45 |
xpAwarded: int
|
| 46 |
|
| 47 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 48 |
def _api_error(status_code: int, error: str, error_type: str) -> HTTPException:
|
| 49 |
return HTTPException(
|
| 50 |
status_code=status_code,
|
|
|
|
| 130 |
"unlockedModules": [],
|
| 131 |
"xpAwarded": 30 if payload.correct else 0,
|
| 132 |
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
routes/jev_routes.py
CHANGED
|
@@ -51,10 +51,12 @@ async def verify_jev(request: JevVerifyRequest) -> JevVerifyResponse:
|
|
| 51 |
status_code=502,
|
| 52 |
detail="JEV verification service unavailable",
|
| 53 |
) from exc
|
|
|
|
|
|
|
| 54 |
verified = bool(verification.get("verified", True))
|
| 55 |
return JevVerifyResponse(
|
| 56 |
verified=verified,
|
| 57 |
-
pCorrect=float(verification.get("pCorrect", 1.0
|
| 58 |
pLeak=float(verification.get("pLeak", 0.0)),
|
| 59 |
-
action=str(verification.get("action", "
|
| 60 |
)
|
|
|
|
| 51 |
status_code=502,
|
| 52 |
detail="JEV verification service unavailable",
|
| 53 |
) from exc
|
| 54 |
+
# This endpoint checks factuality only; claimType does not trigger leak detection.
|
| 55 |
+
# The probability values are conservative fail-open defaults when TypeSafe omits them.
|
| 56 |
verified = bool(verification.get("verified", True))
|
| 57 |
return JevVerifyResponse(
|
| 58 |
verified=verified,
|
| 59 |
+
pCorrect=float(verification.get("pCorrect", 1.0)),
|
| 60 |
pLeak=float(verification.get("pLeak", 0.0)),
|
| 61 |
+
action=str(verification.get("action", "fallback_disabled")),
|
| 62 |
)
|
services/jev_client.py
CHANGED
|
@@ -5,13 +5,56 @@ from __future__ import annotations
|
|
| 5 |
import asyncio
|
| 6 |
import logging
|
| 7 |
import os
|
| 8 |
-
from typing import Any, Callable, Mapping
|
| 9 |
|
| 10 |
logger = logging.getLogger(__name__)
|
| 11 |
|
| 12 |
CLIENT_TIMEOUT_SECONDS = 5.0
|
| 13 |
_REQUEST_LIMIT = asyncio.Semaphore(10)
|
| 14 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 15 |
try:
|
| 16 |
from typesafe_sdk import Choice, Noul, Score, TypeSafeClient
|
| 17 |
except ImportError: # Optional dependency: the backend remains usable offline.
|
|
@@ -72,19 +115,19 @@ PEDAGOGICAL_LEAK = _build_noul(
|
|
| 72 |
BLOOM_MASTERY_SCORE = _build_score()
|
| 73 |
|
| 74 |
|
| 75 |
-
def _fallback_intent() ->
|
| 76 |
return {"action": "fallback_disabled", "choice": "conceptual_confusion"}
|
| 77 |
|
| 78 |
|
| 79 |
-
def _fallback_factuality() ->
|
| 80 |
return {"action": "fallback_disabled", "verified": True}
|
| 81 |
|
| 82 |
|
| 83 |
-
def _fallback_mastery() ->
|
| 84 |
return {"action": "fallback_disabled", "pCorrect": 1.0, "level": 0}
|
| 85 |
|
| 86 |
|
| 87 |
-
def _fallback_safety() ->
|
| 88 |
return {"action": "fallback_disabled", "pLeak": 0.0, "safe": True}
|
| 89 |
|
| 90 |
|
|
@@ -117,7 +160,7 @@ async def _system_one(state: dict[str, Any], questions: dict[str, Any]) -> objec
|
|
| 117 |
async def _call_or_fallback(
|
| 118 |
state: dict[str, Any],
|
| 119 |
questions: dict[str, Any],
|
| 120 |
-
fallback: Callable[[],
|
| 121 |
) -> object | dict[str, Any]:
|
| 122 |
if _api_key() is None:
|
| 123 |
return fallback()
|
|
@@ -130,7 +173,9 @@ async def _call_or_fallback(
|
|
| 130 |
return fallback()
|
| 131 |
|
| 132 |
|
| 133 |
-
async def route_student_intent(
|
|
|
|
|
|
|
| 134 |
"""Classify a student's message into the tutoring intent taxonomy."""
|
| 135 |
response = await _call_or_fallback(
|
| 136 |
{"message": message}, {"intent": INTENT_ROUTING}, _fallback_intent
|
|
@@ -138,16 +183,17 @@ async def route_student_intent(message: str) -> dict[str, Any]:
|
|
| 138 |
if isinstance(response, dict):
|
| 139 |
return response
|
| 140 |
answer = _answers(response).get("intent")
|
| 141 |
-
|
| 142 |
"choice": _answer_field(answer, "choice", "conceptual_confusion"),
|
| 143 |
"confidence": _answer_field(answer, "confidence", 0.0),
|
| 144 |
"probabilities": _answer_field(answer, "probabilities", {}),
|
| 145 |
}
|
|
|
|
| 146 |
|
| 147 |
|
| 148 |
async def verify_lesson_factuality(
|
| 149 |
reference_text: str, generated_text: str
|
| 150 |
-
) -> dict[str, Any]:
|
| 151 |
"""Check generated lesson claims against the supplied reference."""
|
| 152 |
response = await _call_or_fallback(
|
| 153 |
{"reference_text": reference_text, "generated_text": generated_text},
|
|
@@ -158,15 +204,16 @@ async def verify_lesson_factuality(
|
|
| 158 |
return response
|
| 159 |
answer = _answers(response).get("factual_alignment")
|
| 160 |
alignment_probability = _answer_field(answer, "noul", 1.0)
|
| 161 |
-
|
| 162 |
"verified": float(alignment_probability) >= 0.5,
|
| 163 |
"confidence": abs(float(alignment_probability) - 0.5) * 2,
|
| 164 |
}
|
|
|
|
| 165 |
|
| 166 |
|
| 167 |
async def score_student_mastery(
|
| 168 |
student_answers: list[dict[str, Any]], topic: str
|
| 169 |
-
) -> dict[str, Any]:
|
| 170 |
"""Estimate mastery from answers using four Bloom levels."""
|
| 171 |
response = await _call_or_fallback(
|
| 172 |
{"student_answers": student_answers, "topic": topic},
|
|
@@ -177,12 +224,16 @@ async def score_student_mastery(
|
|
| 177 |
return response
|
| 178 |
answer = _answers(response).get("bloom_mastery_score")
|
| 179 |
level = _answer_field(answer, "score", _answer_field(answer, "level", 0))
|
| 180 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
| 181 |
|
| 182 |
|
| 183 |
async def verify_solution_safety(
|
| 184 |
reference_steps: str, student_steps: str
|
| 185 |
-
) -> dict[str, Any]:
|
| 186 |
"""Check whether student-facing steps reveal the final solution."""
|
| 187 |
response = await _call_or_fallback(
|
| 188 |
{"reference_steps": reference_steps, "student_steps": student_steps},
|
|
@@ -193,7 +244,11 @@ async def verify_solution_safety(
|
|
| 193 |
return response
|
| 194 |
answer = _answers(response).get("pedagogical_leak")
|
| 195 |
leak_probability = _answer_field(answer, "noul", 0.0)
|
| 196 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
| 197 |
|
| 198 |
|
| 199 |
__all__ = [
|
|
|
|
| 5 |
import asyncio
|
| 6 |
import logging
|
| 7 |
import os
|
| 8 |
+
from typing import Any, Callable, Mapping, TypedDict
|
| 9 |
|
| 10 |
logger = logging.getLogger(__name__)
|
| 11 |
|
| 12 |
CLIENT_TIMEOUT_SECONDS = 5.0
|
| 13 |
_REQUEST_LIMIT = asyncio.Semaphore(10)
|
| 14 |
|
| 15 |
+
|
| 16 |
+
class IntentSuccess(TypedDict):
|
| 17 |
+
choice: Any
|
| 18 |
+
confidence: Any
|
| 19 |
+
probabilities: Any
|
| 20 |
+
|
| 21 |
+
|
| 22 |
+
class IntentFallback(TypedDict):
|
| 23 |
+
action: Any
|
| 24 |
+
choice: Any
|
| 25 |
+
|
| 26 |
+
|
| 27 |
+
class FactualitySuccess(TypedDict):
|
| 28 |
+
verified: bool
|
| 29 |
+
confidence: float
|
| 30 |
+
|
| 31 |
+
|
| 32 |
+
class FactualityFallback(TypedDict):
|
| 33 |
+
action: str
|
| 34 |
+
verified: bool
|
| 35 |
+
|
| 36 |
+
|
| 37 |
+
class MasterySuccess(TypedDict):
|
| 38 |
+
level: Any
|
| 39 |
+
pCorrect: Any
|
| 40 |
+
|
| 41 |
+
|
| 42 |
+
class MasteryFallback(TypedDict):
|
| 43 |
+
action: Any
|
| 44 |
+
pCorrect: Any
|
| 45 |
+
level: Any
|
| 46 |
+
|
| 47 |
+
|
| 48 |
+
class SafetySuccess(TypedDict):
|
| 49 |
+
pLeak: Any
|
| 50 |
+
safe: Any
|
| 51 |
+
|
| 52 |
+
|
| 53 |
+
class SafetyFallback(TypedDict):
|
| 54 |
+
action: Any
|
| 55 |
+
pLeak: Any
|
| 56 |
+
safe: Any
|
| 57 |
+
|
| 58 |
try:
|
| 59 |
from typesafe_sdk import Choice, Noul, Score, TypeSafeClient
|
| 60 |
except ImportError: # Optional dependency: the backend remains usable offline.
|
|
|
|
| 115 |
BLOOM_MASTERY_SCORE = _build_score()
|
| 116 |
|
| 117 |
|
| 118 |
+
def _fallback_intent() -> IntentFallback:
|
| 119 |
return {"action": "fallback_disabled", "choice": "conceptual_confusion"}
|
| 120 |
|
| 121 |
|
| 122 |
+
def _fallback_factuality() -> FactualityFallback:
|
| 123 |
return {"action": "fallback_disabled", "verified": True}
|
| 124 |
|
| 125 |
|
| 126 |
+
def _fallback_mastery() -> MasteryFallback:
|
| 127 |
return {"action": "fallback_disabled", "pCorrect": 1.0, "level": 0}
|
| 128 |
|
| 129 |
|
| 130 |
+
def _fallback_safety() -> SafetyFallback:
|
| 131 |
return {"action": "fallback_disabled", "pLeak": 0.0, "safe": True}
|
| 132 |
|
| 133 |
|
|
|
|
| 160 |
async def _call_or_fallback(
|
| 161 |
state: dict[str, Any],
|
| 162 |
questions: dict[str, Any],
|
| 163 |
+
fallback: Callable[[], object],
|
| 164 |
) -> object | dict[str, Any]:
|
| 165 |
if _api_key() is None:
|
| 166 |
return fallback()
|
|
|
|
| 173 |
return fallback()
|
| 174 |
|
| 175 |
|
| 176 |
+
async def route_student_intent(
|
| 177 |
+
message: str,
|
| 178 |
+
) -> IntentSuccess | IntentFallback | dict[str, Any]:
|
| 179 |
"""Classify a student's message into the tutoring intent taxonomy."""
|
| 180 |
response = await _call_or_fallback(
|
| 181 |
{"message": message}, {"intent": INTENT_ROUTING}, _fallback_intent
|
|
|
|
| 183 |
if isinstance(response, dict):
|
| 184 |
return response
|
| 185 |
answer = _answers(response).get("intent")
|
| 186 |
+
success: IntentSuccess = {
|
| 187 |
"choice": _answer_field(answer, "choice", "conceptual_confusion"),
|
| 188 |
"confidence": _answer_field(answer, "confidence", 0.0),
|
| 189 |
"probabilities": _answer_field(answer, "probabilities", {}),
|
| 190 |
}
|
| 191 |
+
return success
|
| 192 |
|
| 193 |
|
| 194 |
async def verify_lesson_factuality(
|
| 195 |
reference_text: str, generated_text: str
|
| 196 |
+
) -> FactualitySuccess | FactualityFallback | dict[str, Any]:
|
| 197 |
"""Check generated lesson claims against the supplied reference."""
|
| 198 |
response = await _call_or_fallback(
|
| 199 |
{"reference_text": reference_text, "generated_text": generated_text},
|
|
|
|
| 204 |
return response
|
| 205 |
answer = _answers(response).get("factual_alignment")
|
| 206 |
alignment_probability = _answer_field(answer, "noul", 1.0)
|
| 207 |
+
success: FactualitySuccess = {
|
| 208 |
"verified": float(alignment_probability) >= 0.5,
|
| 209 |
"confidence": abs(float(alignment_probability) - 0.5) * 2,
|
| 210 |
}
|
| 211 |
+
return success
|
| 212 |
|
| 213 |
|
| 214 |
async def score_student_mastery(
|
| 215 |
student_answers: list[dict[str, Any]], topic: str
|
| 216 |
+
) -> MasterySuccess | MasteryFallback | dict[str, Any]:
|
| 217 |
"""Estimate mastery from answers using four Bloom levels."""
|
| 218 |
response = await _call_or_fallback(
|
| 219 |
{"student_answers": student_answers, "topic": topic},
|
|
|
|
| 224 |
return response
|
| 225 |
answer = _answers(response).get("bloom_mastery_score")
|
| 226 |
level = _answer_field(answer, "score", _answer_field(answer, "level", 0))
|
| 227 |
+
success: MasterySuccess = {
|
| 228 |
+
"level": level,
|
| 229 |
+
"pCorrect": _answer_field(answer, "confidence", 0.0),
|
| 230 |
+
}
|
| 231 |
+
return success
|
| 232 |
|
| 233 |
|
| 234 |
async def verify_solution_safety(
|
| 235 |
reference_steps: str, student_steps: str
|
| 236 |
+
) -> SafetySuccess | SafetyFallback | dict[str, Any]:
|
| 237 |
"""Check whether student-facing steps reveal the final solution."""
|
| 238 |
response = await _call_or_fallback(
|
| 239 |
{"reference_steps": reference_steps, "student_steps": student_steps},
|
|
|
|
| 244 |
return response
|
| 245 |
answer = _answers(response).get("pedagogical_leak")
|
| 246 |
leak_probability = _answer_field(answer, "noul", 0.0)
|
| 247 |
+
success: SafetySuccess = {
|
| 248 |
+
"pLeak": leak_probability,
|
| 249 |
+
"safe": not bool(leak_probability),
|
| 250 |
+
}
|
| 251 |
+
return success
|
| 252 |
|
| 253 |
|
| 254 |
__all__ = [
|
tests/test_fun_modules_routes.py
CHANGED
|
@@ -129,15 +129,26 @@ def test_mastery_record_handles_incorrect_answer():
|
|
| 129 |
assert response.json()["xpAwarded"] == 0
|
| 130 |
|
| 131 |
|
| 132 |
-
def
|
| 133 |
-
|
| 134 |
-
"
|
| 135 |
-
|
| 136 |
-
|
| 137 |
-
|
| 138 |
-
|
| 139 |
-
|
| 140 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 141 |
|
| 142 |
assert response.status_code == 200
|
| 143 |
assert set(response.json()) == {"verified", "pCorrect", "pLeak", "action"}
|
|
@@ -153,18 +164,31 @@ def test_jev_verify_fails_open_without_secret():
|
|
| 153 |
)
|
| 154 |
|
| 155 |
assert response.status_code == 200
|
| 156 |
-
assert response.json()
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 157 |
|
| 158 |
|
| 159 |
-
def
|
| 160 |
-
|
| 161 |
-
"
|
| 162 |
-
|
| 163 |
-
)
|
|
|
|
|
|
|
|
|
|
|
|
|
| 164 |
|
| 165 |
assert response.status_code == 200
|
| 166 |
-
assert response.json()
|
| 167 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
| 168 |
|
| 169 |
|
| 170 |
def test_errors_use_identical_keys_and_never_return_secret():
|
|
|
|
| 129 |
assert response.json()["xpAwarded"] == 0
|
| 130 |
|
| 131 |
|
| 132 |
+
def test_jev_verify_returns_canonical_contract():
|
| 133 |
+
with patch(
|
| 134 |
+
"routes.jev_routes.verify_lesson_factuality",
|
| 135 |
+
new=AsyncMock(
|
| 136 |
+
return_value={
|
| 137 |
+
"verified": True,
|
| 138 |
+
"pCorrect": 0.95,
|
| 139 |
+
"pLeak": 0.05,
|
| 140 |
+
"action": "verified",
|
| 141 |
+
}
|
| 142 |
+
),
|
| 143 |
+
):
|
| 144 |
+
response = client.post(
|
| 145 |
+
"/api/jev/verify",
|
| 146 |
+
json={
|
| 147 |
+
"referenceText": "A linear function has a constant rate of change.",
|
| 148 |
+
"generatedText": "A linear function has a constant rate of change.",
|
| 149 |
+
"claimType": "definition",
|
| 150 |
+
},
|
| 151 |
+
)
|
| 152 |
|
| 153 |
assert response.status_code == 200
|
| 154 |
assert set(response.json()) == {"verified", "pCorrect", "pLeak", "action"}
|
|
|
|
| 164 |
)
|
| 165 |
|
| 166 |
assert response.status_code == 200
|
| 167 |
+
assert response.json() == {
|
| 168 |
+
"verified": True,
|
| 169 |
+
"pCorrect": 1.0,
|
| 170 |
+
"pLeak": 0.0,
|
| 171 |
+
"action": "fallback_disabled",
|
| 172 |
+
}
|
| 173 |
|
| 174 |
|
| 175 |
+
def test_jev_verify_is_factuality_only_for_claim_types():
|
| 176 |
+
with patch(
|
| 177 |
+
"routes.jev_routes.verify_lesson_factuality",
|
| 178 |
+
new=AsyncMock(return_value={"action": "fallback_disabled", "verified": True}),
|
| 179 |
+
):
|
| 180 |
+
response = client.post(
|
| 181 |
+
"/api/jev/verify",
|
| 182 |
+
json={"referenceText": "linear function", "generatedText": "quadratic function", "claimType": "hint"},
|
| 183 |
+
)
|
| 184 |
|
| 185 |
assert response.status_code == 200
|
| 186 |
+
assert response.json() == {
|
| 187 |
+
"verified": True,
|
| 188 |
+
"pCorrect": 1.0,
|
| 189 |
+
"pLeak": 0.0,
|
| 190 |
+
"action": "fallback_disabled",
|
| 191 |
+
}
|
| 192 |
|
| 193 |
|
| 194 |
def test_errors_use_identical_keys_and_never_return_secret():
|