Spaces:
Running on Zero
Running on Zero
Download app.py from AngeloUNIMI/document_exam_trainer: direct link, hf CLI and curl.
- Browser
- Download file 40.3 kB
-
https://huggingface.co/spaces/AngeloUNIMI/document_exam_trainer/resolve/main/app.py
- Command line
-
hf download hf://spaces/AngeloUNIMI/document_exam_trainer/app.py
-
curl -L -o app.py https://huggingface.co/spaces/AngeloUNIMI/document_exam_trainer/resolve/main/app.py
40.3 kB
| from __future__ import annotations | |
| # Set these before imports that may load Gradio or torch. | |
| import os | |
| from pathlib import Path | |
| import tempfile | |
| os.environ["GRADIO_SSR_MODE"] = "False" | |
| os.environ["GRADIO_ANALYTICS_ENABLED"] = "False" | |
| os.environ.setdefault("HF_HUB_DISABLE_TELEMETRY", "1") | |
| os.environ.setdefault("TOKENIZERS_PARALLELISM", "false") | |
| os.environ.setdefault("GRADIO_TEMP_DIR", str(Path(tempfile.gettempdir()) / "document-exam-trainer-uploads")) | |
| EDITION = os.getenv("APP_EDITION", "space") | |
| IS_LOCAL = EDITION == "local" | |
| if EDITION not in ("local", "space"): | |
| raise RuntimeError("APP_EDITION must be local or space.") | |
| LOCAL_CONFIG = None | |
| LOCAL_STORE = None | |
| if IS_LOCAL: | |
| from desktop.configuration import configure_environment, DATA, SECRETS | |
| from desktop.storage import LocalStore | |
| LOCAL_CONFIG = configure_environment() | |
| LOCAL_STORE = LocalStore(DATA, SECRETS) | |
| if not LOCAL_STORE.list_users(): | |
| raise RuntimeError("Run the local installation wizard before starting the application.") | |
| if not IS_LOCAL and os.getenv("DEMO_MODE", "0") != "1": | |
| import spaces # MUST be before Gradio/torch/transformers; managed by ZeroGPU. | |
| import hashlib | |
| from contextlib import contextmanager | |
| import html | |
| import uuid | |
| import gradio as gr | |
| from engine import __version__ | |
| from engine.config import SETTINGS | |
| from engine.ingestion import FORMATS, IngestionError | |
| from engine.llm import create_backend | |
| from engine.rendering import md, render_feedback, source_list | |
| from engine.service import TrainerService | |
| from engine.sessions import SessionError | |
| from engine.validation import OutputError | |
| from engine.workers import WorkerError | |
| from engine.webguard import PrivateUploads, UPLOADS, cookie_owner | |
| from starlette.middleware import Middleware | |
| ROOT = Path(__file__).resolve().parent | |
| UPLOAD_ROOT = Path(os.environ["GRADIO_TEMP_DIR"]).resolve() | |
| UPLOAD_ROOT.mkdir(parents=True, exist_ok=True) | |
| if IS_LOCAL: | |
| from desktop.backends import AccountBackend, actor_scope | |
| from desktop.ui_helpers import course_options, progress_text, history_text | |
| BACKEND = AccountBackend(LOCAL_STORE, LOCAL_CONFIG, demo=SETTINGS.demo) | |
| else: | |
| BACKEND = create_backend() # Eager global CUDA placement for ZeroGPU only. | |
| SERVICE = TrainerService(BACKEND) | |
| SERVICE.store.start_reaper() | |
| def owner(request: gr.Request) -> str: | |
| if not request.session_hash: | |
| raise SessionError("Open the app in a browser and refresh the page to create a session.") | |
| browser = cookie_owner(request.headers) | |
| if not browser: | |
| raise SessionError("Session cookies are unavailable. Open the app in its own tab, reload it, and upload again.") | |
| prefix = current_user(request)["id"] + ":" if IS_LOCAL else "" | |
| return prefix + request.session_hash + ":" + browser | |
| def current_user(request): | |
| if not IS_LOCAL: | |
| return None | |
| username = getattr(request, "username", None) if LOCAL_CONFIG.accounts else "local" | |
| record = LOCAL_STORE.user(username) if username else None | |
| if not record: | |
| raise SessionError("Sign in with your local account. An authenticated browser is required.") | |
| return record | |
| def workspace(sid, request): | |
| # Only request.username supplies identity; no account IDs or token arguments | |
| # from a browser may select another user's workspace or credentials. | |
| with SERVICE.store.lease(sid, owner(request)) as session: | |
| yield session | |
| def inference_identity(request): | |
| # A Gradio streaming generator can resume in a different copied context on | |
| # each yield. Bind the account ONLY around a synchronous inference call, | |
| # never across UI yields, so tokens cannot leak or fail to reset. | |
| if IS_LOCAL: | |
| with actor_scope(current_user(request)["id"]): | |
| yield | |
| else: | |
| yield | |
| def uploaded_path(value: str, request: gr.Request) -> Path: | |
| if not value: | |
| raise ValueError("Select an uploaded file first.") | |
| path = Path(str(value)).resolve() | |
| # Do not accept arbitrary server paths, URLs, or paths belonging to the | |
| # app source, model cache, another private workspace, or the host filesystem. | |
| if not path.is_relative_to(UPLOAD_ROOT) or not path.is_file() or Path(str(value)).is_symlink(): | |
| raise ValueError("The temporary upload is unavailable. Please upload it again.") | |
| if not UPLOADS.owns(cookie_owner(request.headers), path): | |
| raise ValueError("This upload is not available in your browser session. Please upload it again.") | |
| return path | |
| def error_text(exc: Exception) -> str: | |
| code = uuid.uuid4().hex[:8] | |
| print(f"[Request failed] id={code} type={type(exc).__name__}", flush=True) | |
| if isinstance(exc, (ValueError, SessionError, OutputError, IngestionError, WorkerError, TimeoutError)): | |
| return "**Could not complete this step:** " + md(str(exc)) | |
| # Surface common platform failures without leaking prompts, answers or paths. | |
| message = str(exc).lower() | |
| if "quota" in message: | |
| return "**GPU quota unavailable.** Wait for your Hugging Face quota to reset. Uploading and transcription do not request ZeroGPU." | |
| if "duration" in message or "timeout" in message or "timed out" in message: | |
| return "**The GPU task exceeded its time allowance.** Try a smaller topic/answer, or ask the host to adjust the task duration setting." | |
| if "cuda" in message or "gpu" in message: | |
| return "**The GPU worker was unavailable.** Your documents remain in this session; retry shortly." | |
| return f"**This step failed.** Please retry. Diagnostic reference: `{code}`." | |
| CSS = """ | |
| .gradio-container { max-width: 1180px !important; margin: auto; } | |
| .step-note { color:var(--body-text-color-subdued); } | |
| #materials-status, #question-status, #speech-status { min-height: 28px; } | |
| #feedback { line-height:1.65; } | |
| """ | |
| with gr.Blocks(title=SETTINGS.app_title, analytics_enabled=False, delete_cache=(600, 3600), fill_width=False) as demo: | |
| token = gr.State(value=SERVICE.store.new_token, delete_callback=SERVICE.store.drop) | |
| gr.Markdown(f"# {SETTINGS.app_title}\nUpload your material. Answer in writing or orally. Discover what your answer is missing.") | |
| if SETTINGS.demo: | |
| gr.Markdown("**DEMO / TEST MODE:** No language models or semantic embeddings are running. Questions and feedback are deterministic test fixtures, NOT educational evaluations.") | |
| if IS_LOCAL: | |
| gr.Markdown("**Local Docker edition.** Your courses and practice history are saved in Docker volumes. CPU mode keeps inference local. " | |
| "Remote HF mode sends selected text excerpts and answer text to the configured Space, using your own token.") | |
| else: | |
| gr.Markdown("**Hosted demo.** Sessions are temporary. Open **Download & install** for the local Docker edition with accounts and saved learning history.") | |
| gr.Markdown("> **How feedback works:** There are no grades. The trainer reports **Essential**, **Important**, and **Minor** omissions relative to the question. " | |
| "Primary documents define what is required; supporting references cannot add requirements.") | |
| with gr.Tabs(): | |
| with gr.Tab("1. Materials"): | |
| gr.Markdown("### Build your study collection\nUpload primary material to start immediately. Supporting references are optional.") | |
| with gr.Row(): | |
| primary = gr.File(label="Primary / examinable material", file_count="multiple", type="filepath", | |
| file_types=sorted(FORMATS | {".zip"})) | |
| supporting = gr.File(label="Supporting references (optional)", file_count="multiple", type="filepath", | |
| file_types=sorted(FORMATS | {".zip"})) | |
| gr.Markdown("PDF, PPTX, DOCX, TXT, Markdown, or a ZIP containing these files. Topics are inferred from document titles and sections. " | |
| "PDF references refer to **PDF pages**, not guessed slide numbers. Image-only text and diagrams are not interpreted.", elem_classes="step-note") | |
| course_name = gr.Textbox(label="Course name (saved locally)", value="My course", visible=IS_LOCAL, max_lines=1) | |
| consent = gr.Checkbox(label="I have permission to process these documents on this application host.", value=False) | |
| with gr.Row(): | |
| build_btn = gr.Button("Process documents", variant="primary") | |
| clear_btn = gr.Button("Clear my session", variant="secondary") | |
| materials_status = gr.Markdown("Upload at least one primary document.", elem_id="materials-status") | |
| manifest = gr.Dataframe(headers=["Document", "Role", "Pages / units", "Text chunks"], | |
| datatype=["str", "str", "number", "number"], interactive=False, label="Collection summary") | |
| extraction_notes = gr.Markdown("") | |
| with gr.Tab("2. Practice"): | |
| with gr.Row(): | |
| topic = gr.Dropdown(choices=[], label="Topic / section", interactive=True, scale=3) | |
| style = gr.Dropdown(choices=["General description", "Definitions and properties", "Procedure / operation", "Limitations and extensions"], | |
| value="General description", label="Question style", scale=2) | |
| focus = gr.Textbox(label="Optional focus within this topic", placeholder="For example, definition and operation. Leave blank for a general question.", max_lines=2) | |
| with gr.Row(): | |
| generate_btn = gr.Button("Generate question", variant="primary", interactive=False) | |
| followup_btn = gr.Button("Practise one identified gap", interactive=False) | |
| question_status = gr.Markdown("Process your documents in the Materials tab first.", elem_id="question-status") | |
| question = gr.Textbox(label="Question", interactive=False, lines=3, buttons=["copy"]) | |
| with gr.Accordion("Optional oral input: record or upload audio", open=False): | |
| gr.Markdown("You can answer directly in writing below, or record an oral answer and select **Transcribe recording**. " | |
| "The transcription is inserted into the answer box and can be reviewed or edited before evaluation. " | |
| "Only content is checked, not pronunciation or speaking style.") | |
| audio = gr.Audio(label=f"Your recording (up to {SETTINGS.max_audio_seconds // 60} minutes)", | |
| sources=["microphone", "upload"], type="filepath", format=None, buttons=[]) | |
| language = gr.Dropdown(choices=[("English", "en"), ("Auto-detect", "auto"), ("Italian", "it"), | |
| ("French", "fr"), ("German", "de"), ("Spanish", "es"), | |
| ("Portuguese", "pt"), ("Dutch", "nl"), ("Arabic", "ar"), ("Chinese", "zh"), ("Japanese", "ja")], | |
| value="en", label="Spoken language") | |
| vocab = gr.Textbox(label="Optional technical terms for transcription", placeholder="Names or technical terms only; do not enter a model answer.", lines=1) | |
| transcribe_btn = gr.Button("Transcribe recording", variant="secondary") | |
| speech_status = gr.Markdown("Speech recognition runs on this host CPU, without a ZeroGPU request.", elem_id="speech-status") | |
| transcript = gr.Textbox(label="Your answer", interactive=True, placeholder="Type your answer here, or use the optional recording section above to transcribe a spoken answer.", lines=9, max_lines=20, buttons=["copy"]) | |
| support_hints = gr.Checkbox(label="Add supporting explanations when references are available (uses an extra inference request)", value=False) | |
| evaluate_btn = gr.Button("Check answer for omissions", variant="primary", interactive=False) | |
| feedback = gr.Markdown("", elem_id="feedback", buttons=["copy"]) | |
| with gr.Accordion("Primary passages used for this question", open=False): | |
| primary_sources = gr.Markdown("Generate a question to see its source passages here.") | |
| if IS_LOCAL: | |
| with gr.Tab("3. Saved courses & progress"): | |
| gr.Markdown("### Your local learning history\nOnly courses belonging to the signed-in local account are listed. " | |
| "After processing, courses are saved automatically. Answer text and omission checks are saved after evaluation; raw audio is not retained in the course library.") | |
| saved_courses = gr.Dropdown(label="Saved course", choices=[], interactive=True) | |
| with gr.Row(): | |
| refresh_courses_btn = gr.Button("Refresh list") | |
| load_course_btn = gr.Button("Load course", variant="primary") | |
| progress_btn = gr.Button("Show learning history") | |
| recommend_btn = gr.Button("Practise a recurring gap") | |
| confirm_delete = gr.Checkbox(label="Delete this saved course and all of its practice history", value=False) | |
| delete_course_btn = gr.Button("Delete saved course", variant="stop") | |
| library_status = gr.Markdown("") | |
| learning_progress = gr.Markdown("") | |
| learning_history = gr.Markdown("") | |
| with gr.Tab("4. Account & inference"): | |
| account_status = gr.Markdown("") | |
| backend_choice = gr.Radio(choices=[("Installation default", "default"), ("Local CPU", "local_cpu"), ("Remote Hugging Face Space", "remote_hf")], value="default", label="Inference backend for this account") | |
| gr.Markdown("Document indexing and transcription stay local in either mode. CPU mode can be slow and uses a smaller model by default. " | |
| "Remote mode sends primary excerpts, questions, rubrics and answer text, and optional supporting excerpts, to Hugging Face. " | |
| "The Space operator/platform can access that data during processing. Original files and audio are not sent to the inference API.") | |
| remote_consent = gr.Checkbox(label="I agree to send these text excerpts and my answer text to the configured Hugging Face Space when remote inference is selected.", value=False) | |
| hf_token = gr.Textbox(label="My Hugging Face token (leave blank to keep the saved token)", type="password", max_lines=1) | |
| erase_token = gr.Checkbox(label="Remove my saved Hugging Face token", value=False) | |
| save_settings_btn = gr.Button("Save my settings", variant="primary") | |
| gr.Markdown("Tokens are encrypted in the local database with a key in a separate Docker volume. Your installation administrator can still access them. " | |
| "No developer token is bundled. Use a personal token with the minimum access needed; do not use your deployment/write token. " | |
| "Local passwords and HF tokens are different credentials. Users do not need an HF token for local CPU inference.") | |
| old_password = gr.Textbox(label="Current local password", type="password", visible=LOCAL_CONFIG.accounts) | |
| new_password = gr.Textbox(label="New local password (at least 12 characters)", type="password", visible=LOCAL_CONFIG.accounts) | |
| repeat_password = gr.Textbox(label="Repeat new password", type="password", visible=LOCAL_CONFIG.accounts) | |
| password_btn = gr.Button("Change my password", visible=LOCAL_CONFIG.accounts) | |
| account_action_status = gr.Markdown("") | |
| if LOCAL_CONFIG.accounts: | |
| gr.Button("Sign out", link="/logout") | |
| else: | |
| with gr.Tab("Download & install"): | |
| from desktop.configuration import DEFAULT_SPACE, repo_id | |
| public_repo = repo_id(os.getenv("SPACE_ID", os.getenv("PUBLIC_SPACE_ID", DEFAULT_SPACE))) | |
| release_url = f"https://huggingface.co/spaces/{public_repo}/resolve/main/downloads/document_exam_trainer_docker.zip?download=true" | |
| gr.Markdown("## Run the trainer on your own computer\nThe Docker edition adds local accounts, saved courses, practice history and recurring-gap practice. " | |
| "Choose **local CPU** or **remote HF inference** during setup, and switch per account later. " | |
| "The download contains the Docker build recipe and application source, not a prebuilt multi-gigabyte image or model weights.") | |
| with gr.Row(): | |
| gr.Button("Download for Windows (Docker)", link=release_url, variant="primary") | |
| gr.Button("Download for Linux (Docker)", link=release_url) | |
| gr.Markdown("**Windows:** extract the archive, then double-click `install_windows.bat`.\n\n" | |
| "**Linux:** extract it, open a terminal there, and run `bash install_linux.sh`.\n\n" | |
| "Setup checks Docker and Compose, offers supported installation options, checks daemon access, builds the image and opens an interactive setup wizard. " | |
| "It does not silently grant administrator privileges or select paid hardware.\n\n" | |
| "**Setup choices:** account login or single-user loopback mode; local usernames/passwords; default CPU or HF backend; " | |
| "per-user HF tokens and explicit remote-processing consent. Open `http://localhost:7860` after setup.\n\n" | |
| "**Keep in mind:** the first build/model downloads require internet. Remote mode still uses the calling HF account's quota. " | |
| "CPU-only quality and speed depend on your model and computer; no GPU or HF quota is required for CPU inference. " | |
| "Data persists in named Docker volumes. Removing the container alone does not delete it.") | |
| with gr.Tab("Help & privacy"): | |
| gr.Markdown(f"""### Practice workflow | |
| Upload primary and optional supporting material, process once, choose a topic, generate a question, then either type your answer directly or record and transcribe it. A transcription can be edited before evaluation. Accent and speaking style are not graded. | |
| ### Source policy | |
| Primary documents define what can be asked. Supporting documents can clarify confirmed gaps but never add requirements. Source checks, semantic comparison and question-specific rubrics are retained. Feedback is not guaranteed complete or correct. Scans, diagrams and image-only formulas are not interpreted; review extraction warnings. | |
| ### Data handling | |
| {'The local edition stores extracted course text, embeddings, answer text and results in Docker volumes, separated by local account. Clearing a session does not delete saved courses; use Saved courses to delete them. SQLite course/history data is not encrypted at rest. HF tokens are encrypted, but the machine administrator can access the encryption key. There is no cloud synchronization.' if IS_LOCAL else 'The hosted demo keeps a temporary workspace for each browser session. There is no shared corpus catalog and no automatic cloud backup. Refreshes or restarts can lose work. The downloadable edition instead supports local accounts and persistent learning history.'} | |
| Original internal document copies are removed after extraction; internal audio copies are removed after transcription. Browser upload caches expire separately. Temporary workspaces expire after {SETTINGS.session_ttl//3600} hours of inactivity. The host administrator can access runtime data. Use only material you are authorized to process. This app contains no model-training pipeline. | |
| ### Microphone and resources | |
| Use HTTPS or localhost, allow microphone access, and open the app directly rather than in an iframe when permissions are blocked. Maximum recording: {SETTINGS.max_audio_seconds} seconds. Transcription is record-then-transcribe, not live streaming. CPU indexing, speech and local LLM jobs can take minutes on slower machines. In remote/hosted mode, generation and evaluation are subject to ZeroGPU queues and the caller's quota. | |
| ### Local administration | |
| The Docker download includes `manage_users_windows.bat` and `manage_users_linux.sh` to add local users, reset passwords and remove accounts. The default published port is bound to **127.0.0.1 only**. Do not expose this local service directly to the public internet; multi-device access requires HTTPS and a reviewed deployment. | |
| Version {__version__}. No numerical grades are assigned. | |
| """) | |
| all_outputs = [primary, supporting, build_btn, clear_btn, materials_status, manifest, extraction_notes, | |
| topic, style, focus, generate_btn, followup_btn, question_status, question, audio, | |
| transcribe_btn, speech_status, transcript, evaluate_btn, feedback, primary_sources, course_name] | |
| if IS_LOCAL: | |
| all_outputs += [saved_courses, library_status, learning_progress, learning_history, confirm_delete] | |
| action_buttons = [build_btn, clear_btn, generate_btn, followup_btn, transcribe_btn, evaluate_btn] | |
| def busy(): | |
| return {c: gr.update(interactive=False) for c in action_buttons + [primary, supporting, topic, style, focus, course_name]} | |
| def ready(session=None): | |
| have_index = session is not None and session.index is not None | |
| have_question = have_index and session.rubric is not None | |
| have_result = have_question and session.last_result is not None | |
| result = {c: gr.update(interactive=True) for c in [build_btn, clear_btn, transcribe_btn, primary, supporting, topic, style, focus]} | |
| result.update({generate_btn: gr.update(interactive=have_index), transcribe_btn: gr.update(interactive=have_question), course_name: gr.update(interactive=True), evaluate_btn: gr.update(interactive=have_question), followup_btn: gr.update(interactive=have_result)}) | |
| return result | |
| def process_documents(primary_files, supporting_files, accepted, sid, requested_name, request: gr.Request, progress=gr.Progress()): | |
| session = None | |
| try: | |
| with workspace(sid, request) as session: | |
| if not accepted: | |
| raise ValueError("Confirm that you have permission to process the documents.") | |
| if not primary_files: | |
| raise ValueError("Upload at least one primary document.") | |
| yield {**busy(), materials_status: "**Processing documents...** Extracting text, then building the semantic index on CPU. The first run downloads the embedding model and can take a few minutes."} | |
| inputs = [(uploaded_path(p, request), "primary") for p in primary_files] | |
| inputs += [(uploaded_path(p, request), "supporting") for p in (supporting_files or [])] | |
| parsed = SERVICE.build(session, inputs, lambda text: progress(None, desc=text)) | |
| if IS_LOCAL: | |
| LOCAL_STORE.save_course(current_user(request)["id"], requested_name or "My course", session, SETTINGS.embedding_model + ("::demo" if SETTINGS.demo else "")) | |
| rows = [[d.filename, d.role.title(), d.units, d.chunks] for d in parsed.documents] | |
| warnings = "\n\n".join("- " + md(t) for t in parsed.warnings[:30]) | |
| if len(parsed.warnings) > 30: | |
| warnings += f"\n\n...and {len(parsed.warnings)-30} additional extraction notices." | |
| yield {**ready(session), materials_status: f"**Ready.** {len(parsed.documents)} documents, {len(parsed.chunks)} text chunks, {len(parsed.topics)} topic/section choices. Open **Practice** to begin.", | |
| manifest: rows, extraction_notes: ("### Extraction notes\n" + warnings) if warnings else "Text extracted successfully. Review the primary passages during practice.", | |
| topic: gr.update(choices=session.index.options(), value="all", interactive=True), | |
| primary: None, supporting: None, question: "", transcript: "", feedback: "", primary_sources: "", | |
| question_status: "Choose a topic and generate a question.", speech_status: "Ready for an optional recording.", | |
| **({saved_courses: gr.update(choices=course_options(LOCAL_STORE,current_user(request)["id"]),value=session.course_id),library_status:"Course saved locally."} if IS_LOCAL else {})} | |
| except Exception as exc: | |
| yield {**ready(session), materials_status: error_text(exc)} | |
| def generate_question(topic_id, selected_style, selected_focus, sid, request: gr.Request, progress=gr.Progress()): | |
| session = None | |
| try: | |
| with workspace(sid, request) as session: | |
| yield {**busy(), question_status: "**Generating a question...** Selecting primary passages, then creating and auditing the hidden rubric."} | |
| with inference_identity(request): | |
| rubric = SERVICE.question(session, topic_id, selected_focus or "", selected_style) | |
| yield {**ready(session), question: rubric.question, transcript: "", audio: None, feedback: "", | |
| primary_sources: source_list(rubric, session.index.by_id), | |
| question_status: "**Question ready.** Type your answer below, or use the optional recording section, then check it for omissions.", speech_status: "Ready for an optional recording."} | |
| except Exception as exc: | |
| yield {**ready(session), question_status: error_text(exc)} | |
| def transcribe_recording(recording, spoken_language, terms, sid, request: gr.Request, progress=gr.Progress()): | |
| session = None | |
| try: | |
| with workspace(sid, request) as session: | |
| if session.rubric is None: | |
| raise ValueError("Generate a question before transcribing an answer.") | |
| path = uploaded_path(recording, request) | |
| yield {**busy(), speech_status: "**Transcribing your recording...** CPU processing; the first use also downloads the speech model."} | |
| result = SERVICE.transcribe(session, path, spoken_language, terms or "", lambda text: progress(None, desc=text)) | |
| text = result["text"] | |
| if len(text) > SETTINGS.max_answer_chars: | |
| raise ValueError("The transcript is too long. Use a shorter recording.") | |
| session.last_result, session.last_key = None, "" | |
| session.transcript = text | |
| session.transcript_question = session.rubric.model_dump_json() | |
| yield {**ready(session), transcript: text, audio: None, feedback: "", speech_status: f"**Transcription ready** ({result['seconds']} seconds of audio; language: {md(result['language'])}). Review or edit the answer text if needed, then check it for omissions."} | |
| except Exception as exc: | |
| yield {**ready(session), speech_status: error_text(exc)} | |
| def evaluate(answer_text, extra_hints, sid, request: gr.Request, progress=gr.Progress()): | |
| session = None | |
| try: | |
| with workspace(sid, request) as session: | |
| yield {**busy(), feedback: "**Evaluating your answer...** Preparing the primary evidence and checking the frozen rubric. No grade will be assigned."} | |
| if session.rubric is None: | |
| raise ValueError("Generate a question first.") | |
| answer = (answer_text or "").strip() | |
| if not answer: | |
| raise ValueError("Type an answer or transcribe a recording before evaluation.") | |
| if len(answer) > SETTINGS.max_answer_chars: | |
| raise ValueError("The answer is too long. Shorten it before evaluation.") | |
| session.last_result, session.last_key = None, "" | |
| session.transcript = answer | |
| session.transcript_question = session.rubric.model_dump_json() | |
| with inference_identity(request): | |
| result = SERVICE.evaluate(session, answer, extra_hints, lambda text: progress(None, desc=text)) | |
| if IS_LOCAL: | |
| uid = current_user(request)["id"] | |
| LOCAL_STORE.save_attempt(uid, session, BACKEND.mode(uid)) | |
| yield {**ready(session), feedback: render_feedback(session.rubric, result, session.index.by_id)} | |
| except Exception as exc: | |
| yield {**ready(session), feedback: error_text(exc)} | |
| def followup(sid, request: gr.Request): | |
| session = None | |
| try: | |
| with workspace(sid, request) as session: | |
| rubric = SERVICE.followup(session) | |
| return {**ready(session), question: rubric.question, transcript: "", audio: None, feedback: "", | |
| primary_sources: source_list(rubric, session.index.by_id), | |
| question_status: "**Follow-up ready.** This targets one previously identified gap. No question-generation GPU request was needed."} | |
| except Exception as exc: | |
| return {**ready(session), question_status: error_text(exc)} | |
| def clear_session(sid, request: gr.Request): | |
| session = None | |
| try: | |
| with workspace(sid, request) as session: | |
| SERVICE.store.clear_locked(session) | |
| UPLOADS.revoke(cookie_owner(request.headers)) | |
| return {**ready(session), primary: None, supporting: None, audio: None, question: "", transcript: "", feedback: "", | |
| manifest: [], extraction_notes: "", topic: gr.update(choices=[], value=None), primary_sources: "", | |
| materials_status: "**Session cleared.** Temporary workspace removed." + (" Saved courses and learning history remain; delete them in Saved courses." if IS_LOCAL else " Upload-cache cleanup follows the retention policy."), | |
| question_status: "Upload new primary documents to begin again.", speech_status: "Ready for an optional recording."} | |
| except Exception as exc: | |
| return {**ready(session), materials_status: error_text(exc)} | |
| if IS_LOCAL: | |
| def account_view(request: gr.Request): | |
| u = current_user(request) | |
| mode = BACKEND.mode(u["id"]) | |
| status = (f"**Signed in:** {md(u['username'])}. **Backend:** {md(mode)}. " | |
| f"**Remote Space:** {md(LOCAL_CONFIG.remote_space)}. " | |
| + ("A personal HF token is saved." if u['hf_token'] else "No HF token saved.")) | |
| return {account_status:status, backend_choice:u['backend'],remote_consent:bool(u['remote_consent']), | |
| saved_courses:gr.update(choices=course_options(LOCAL_STORE,u['id']))} | |
| def save_preferences(choice, approved, new_token, remove, request: gr.Request): | |
| try: | |
| u = current_user(request) | |
| LOCAL_STORE.preferences(u['id'],choice,approved,new_token or '',remove) | |
| return {**account_view(request),hf_token:'',erase_token:False,account_action_status:'**Settings saved.** Changes apply to your next request.'} | |
| except Exception as exc: | |
| return {hf_token:'',account_action_status:error_text(exc)} | |
| def password_change(old,new,repeat,request: gr.Request): | |
| try: | |
| if not LOCAL_CONFIG.accounts: | |
| raise ValueError('Password login is disabled for this installation.') | |
| if new != repeat: | |
| raise ValueError('New passwords do not match.') | |
| LOCAL_STORE.change_password(current_user(request)['id'],old,new) | |
| text='**Password changed.** Already logged-in sessions remain active until logout or restart.' | |
| except Exception as exc: | |
| text=error_text(exc) | |
| return {old_password:'',new_password:'',repeat_password:'',account_action_status:text} | |
| def saved_course_load(cid,sid,request: gr.Request): | |
| session=None | |
| try: | |
| with workspace(sid,request) as session: | |
| uid=current_user(request)['id'] | |
| parsed=LOCAL_STORE.load_course(uid,cid,session,SETTINGS.embedding_model + ('::demo' if SETTINGS.demo else '')) | |
| name=LOCAL_STORE.owned_course(uid,cid)['name'] | |
| return {**ready(session),topic:gr.update(choices=session.index.options(),value='all'),course_name:name, | |
| manifest:[[d.filename,d.role.title(),d.units,d.chunks] for d in parsed.documents], | |
| materials_status:'**Saved course loaded.** Open Practice to begin.',library_status:'**Course loaded.**',question:'',transcript:'',audio:None,feedback:'',primary_sources:'', | |
| question_status:'Choose a topic and generate a question.'} | |
| except Exception as exc: | |
| return {**ready(session),library_status:error_text(exc)} | |
| def show_history(cid,request: gr.Request): | |
| try: | |
| uid=current_user(request)['id'] | |
| return {learning_progress:progress_text(LOCAL_STORE,uid,cid),learning_history:history_text(LOCAL_STORE,uid,cid),library_status:''} | |
| except Exception as exc: | |
| return {library_status:error_text(exc)} | |
| def recommend_gap(cid,sid,request: gr.Request,progress=gr.Progress()): | |
| session=None | |
| try: | |
| with workspace(sid,request) as session: | |
| uid=current_user(request)['id'] | |
| if session.course_id != cid: | |
| LOCAL_STORE.load_course(uid,cid,session,SETTINGS.embedding_model + ('::demo' if SETTINGS.demo else '')) | |
| gaps=[r for r in LOCAL_STORE.progress(uid,cid) if r['latest'] in ('missing','partial','incorrect')] | |
| if not gaps: | |
| raise ValueError('No recent confirmed gaps are available. Practise a new topic first.') | |
| target=gaps[0] | |
| yield {**busy(),library_status:'**Preparing personalized practice...** This uses a question-generation request.',question_status:'Generating a question about a recurring gap...'} | |
| with inference_identity(request): | |
| rubric=SERVICE.question(session,target['topic_id'],target['concept'],'General description') | |
| yield {**ready(session),question:rubric.question,transcript:'',audio:None,feedback:'',primary_sources:source_list(rubric,session.index.by_id), | |
| topic:gr.update(choices=session.index.options(),value=target['topic_id']), | |
| library_status:'**Personalized question ready.** Open Practice and type or record your answer.',question_status:'**Question ready.** Focus: '+md(target['concept'])} | |
| except Exception as exc: | |
| yield {**ready(session),library_status:error_text(exc)} | |
| def delete_saved(cid,confirmed,sid,request: gr.Request): | |
| session=None | |
| try: | |
| if not confirmed: | |
| raise ValueError('Tick the deletion confirmation first.') | |
| with workspace(sid,request) as session: | |
| uid=current_user(request)['id'] | |
| LOCAL_STORE.delete_course(uid,cid) | |
| # Clear this session so its next evaluation cannot recreate a deleted course. | |
| SERVICE.store.clear_locked(session) | |
| return {**ready(session),saved_courses:gr.update(choices=course_options(LOCAL_STORE,uid),value=None), | |
| library_status:'**Saved course and history deleted.** Other open browser tabs should be refreshed.', | |
| learning_progress:'',learning_history:'',confirm_delete:False,question:'',transcript:'',audio:None,feedback:'',primary_sources:'',manifest:[], | |
| topic:gr.update(choices=[],value=None),materials_status:'Upload documents or load another saved course.',question_status:'No active course.'} | |
| except Exception as exc: | |
| return {**ready(session),library_status:error_text(exc)} | |
| account_outputs=[account_status,backend_choice,remote_consent,saved_courses,hf_token,erase_token,account_action_status] | |
| demo.load(account_view,outputs=account_outputs,api_visibility='private') | |
| save_settings_btn.click(save_preferences,[backend_choice,remote_consent,hf_token,erase_token],account_outputs,api_visibility='private',concurrency_id='llm') | |
| password_btn.click(password_change,[old_password,new_password,repeat_password],[old_password,new_password,repeat_password,account_action_status],api_visibility='private') | |
| refresh_courses_btn.click(account_view,outputs=account_outputs,api_visibility='private') | |
| load_course_btn.click(saved_course_load,[saved_courses,token],all_outputs,api_visibility='private',concurrency_id='llm') | |
| progress_btn.click(show_history,[saved_courses],[learning_progress,learning_history,library_status],api_visibility='private') | |
| recommend_btn.click(recommend_gap,[saved_courses,token],all_outputs,api_visibility='private',concurrency_id='llm') | |
| delete_course_btn.click(delete_saved,[saved_courses,confirm_delete,token],all_outputs,api_visibility='private',concurrency_id='llm') | |
| else: | |
| # A separate, stateless endpoint for Docker clients. No original files, | |
| # audio, local-account IDs or HF tokens are accepted as task parameters. | |
| rpc_task=gr.Textbox(visible=False) | |
| rpc_payload=gr.Textbox(visible=False) | |
| rpc_output=gr.Textbox(visible=False) | |
| rpc_button=gr.Button(visible=False) | |
| def infer_v1(task,payload_json,request: gr.Request): | |
| from engine.rpc import validate_payload, make_response | |
| data=validate_payload(task,payload_json) | |
| return make_response(task,BACKEND.call(task,data)) | |
| rpc_button.click(infer_v1,[rpc_task,rpc_payload],rpc_output,api_name='trainer_infer_v1',api_visibility='public',concurrency_id='llm',concurrency_limit=1) | |
| private_event = dict(outputs=all_outputs, api_visibility="private", show_progress="full", trigger_mode="once") | |
| build_btn.click(process_documents, [primary, supporting, consent, token, course_name], concurrency_id="cpu-heavy", concurrency_limit=1, | |
| show_progress_on=[materials_status], **private_event) | |
| generate_btn.click(generate_question, [topic, style, focus, token], concurrency_id="llm", concurrency_limit=1, | |
| show_progress_on=[question_status], **private_event) | |
| transcribe_btn.click(transcribe_recording, [audio, language, vocab, token], concurrency_id="cpu-heavy", concurrency_limit=1, | |
| show_progress_on=[speech_status], **private_event) | |
| evaluate_btn.click(evaluate, [transcript, support_hints, token], concurrency_id="llm", concurrency_limit=1, | |
| show_progress_on=[feedback], **private_event) | |
| followup_btn.click(followup, [token], concurrency_id="llm", concurrency_limit=1, **private_event) | |
| clear_btn.click(clear_session, [token], concurrency_id="llm", concurrency_limit=1, **private_event) | |
| def launch_app(): | |
| # HTTP middleware protects uploads without breaking input preprocessing. | |
| # Only the uploading browser can replay its recording. Workspace paths are also blocked explicitly. | |
| demo.queue(default_concurrency_limit=1, max_size=32).launch( | |
| server_name="0.0.0.0", ssr_mode=False, show_error=False, theme=gr.themes.Soft(), css=CSS, | |
| max_file_size=SETTINGS.max_file_bytes, blocked_paths=[str(SETTINGS.root)] + ([str(DATA),str(SECRETS)] if IS_LOCAL else []), | |
| app_kwargs={"middleware": [Middleware(PrivateUploads)]}, | |
| auth=LOCAL_STORE.authenticate if IS_LOCAL and LOCAL_CONFIG.accounts else None, | |
| auth_message="Sign in with your local trainer account, not your Hugging Face account." if IS_LOCAL else None, | |
| share=False, | |
| state_session_capacity=SETTINGS.max_sessions * 2, enable_monitoring=False, mcp_server=False, | |
| footer_links=["gradio"], | |
| ) | |
| if __name__ == "__main__": | |
| launch_app() | |