AnatoBot / app.py
stevafernandes's picture
Upload 10 files
191750c verified
Raw History Blame Contribute Delete
103 kB
"""Creighton Anatomy Learning Platform.
A Gradio app for Hugging Face Spaces. Every answer is grounded in the lecture
PDFs found in the Lectures folder and generated with the Gemini API; when the
slides do not cover a request, reference articles from TeachMeAnatomy and
Kenhub are used instead and named as the source. The tutor never answers from
its own knowledge. Slide images shown in the viewer are rendered directly from
the PDFs. Accounts, chat history, and usage analytics are stored in Supabase.
"""
from __future__ import annotations
import atexit
import base64
import io
import logging
import math
import os
import queue
import re
import shutil
import tempfile
import threading
import time
import uuid
import zipfile
from collections import Counter, OrderedDict
from concurrent.futures import ThreadPoolExecutor
from dataclasses import dataclass, field
from pathlib import Path
import fitz # PyMuPDF
import gradio as gr
from google import genai
from google.genai import errors as genai_errors
from google.genai import types as genai_types
from PIL import Image
from pydantic import BaseModel, ValidationError
import web_sources
from supabase_store import EXISTS, SupabaseStore, is_valid_email, is_valid_user_id, normalize_email, normalize_user_id
logging.basicConfig(level=logging.WARNING, format="%(asctime)s %(levelname)s %(message)s")
log = logging.getLogger("anatomy")
log.setLevel(logging.INFO)
# --------------------------------------------------------------------------- #
# Configuration
# --------------------------------------------------------------------------- #
APP_DIR = Path(__file__).resolve().parent
def _has_pdfs(folder: Path) -> bool:
return folder.is_dir() and any(p.is_file() for p in folder.glob("*.pdf"))
def find_lectures(app_dir: Path, configured: Path) -> Path:
"""Folder holding the lecture PDFs; unpacks a Lectures zip next to app.py when needed."""
if _has_pdfs(configured):
return configured
archives = sorted(app_dir.glob("*.zip"), key=lambda p: (p.name.lower() != "lectures.zip", p.name))
for archive in archives:
target = app_dir / "_lectures"
if not _has_pdfs(target) and not any(_has_pdfs(d) for d in target.rglob("*") if d.is_dir()):
log.info("Extracting %s", archive.name)
with zipfile.ZipFile(archive) as zf:
zf.extractall(target)
for pdf in sorted(target.rglob("*.pdf")):
if "__MACOSX" in pdf.parts or pdf.name.startswith("._"):
continue
return pdf.parent
return configured
LECTURES_DIR = find_lectures(APP_DIR, Path(os.environ.get("LECTURES_DIR", APP_DIR / "Lectures")))
LOGO_NAMES = ("creighton-bluejays@logotyp.us.png", "resize.webp") # first one found is used
def find_logo(*folders: Path) -> Path | None:
"""The Creighton logo, looked up in the lecture folder first and then next to app.py."""
for folder in folders:
for name in LOGO_NAMES:
if (folder / name).is_file():
return folder / name
return None
LOGO_PATH = find_logo(LECTURES_DIR, APP_DIR)
GEMINI_API_KEY = os.environ.get("GEMINI_API_KEY") or os.environ.get("GOOGLE_API_KEY")
GEMINI_MODEL = os.environ.get("GEMINI_MODEL", "gemini-2.5-flash")
# Tried in order if the configured model is not found for this API key.
FALLBACK_MODELS = ["gemini-2.5-flash", "gemini-3.8-flash", "gemini-3.5-flash"]
GEMINI_TIMEOUT_MS = 120_000
MAX_HISTORY_TURNS = 6 # prior user/model exchanges sent to Gemini
MAX_CONTEXT_SLIDES = 8 # slide images attached to a request
MAX_VIEWER_SLIDES = 6 # upper bound on slides the model may pick for the viewer
MAX_SESSIONS = 1000 # browser sessions kept in memory
RENDER_WIDTH = 1000 # pixel width of slide images sent to Gemini
RENDER_CACHE_SIZE = 200
VIEWER_WIDTH = 2000 # pixel width of slides shown in the viewer (sharp on high-density screens and in fullscreen)
VIEWER_QUALITY = 90 # WebP quality of viewer slides
DIAGRAM_MAX_CHARS = 25 # slides with less text are title-only figure slides
TITLE_WEIGHT = 0.25 # weight of a lecture-title match in retrieval
LECTURE_FOCUS = 0.6 # slide images come only from lectures scoring at least this share of the best lecture
FALLBACK_MODES = ("ask", "quiz") # answered from the reference sites when the slides do not cover them
HEARTBEAT_SECONDS = 30 # how often an open tab's time on the site is written to Supabase
SYNC_SECONDS = 15 # how often an open tab checks that it shows the latest chat (recovers a dropped answer)
RESTORE_MESSAGES = 200 # stored messages shown again when a user signs in
ANSWER_WAIT_SECONDS = 30 # how long a reloaded page waits for an answer its old tab is still writing (later: on_sync)
CU_BLUE = "#005CA9"
CU_NAVY = "#00235D"
CU_LIGHT_BLUE = "#6CADDE"
CU_LIGHT_GRAY = "#C8C8C8"
CU_TINT = "#EAF3FB"
FILENAME_RE = re.compile(r"^Lecture #(\d+) - (.+)\.pdf$", re.IGNORECASE)
SECTION_TITLE_RE = re.compile(
r"^(?:summary|review|questions?|fyi|practical application|laugh break|challenge|title|objectives?|thank you|the end)[\s!?.:]*$",
re.IGNORECASE,
)
_RANGE = r"\d+(?:\s*(?:-|–|to)\s*\d+)?"
_SEP = r"\s*(?:,|;|&|,?\s*(?:and|or))\s*(?:slides?\s+)?" # "slides 3, 5", "slides 3 and 5", "slide 3, slide 5"
CITATION_RE = re.compile(rf"\bLecture\s+(\d+)\s*,?\s*slides?\s+({_RANGE}(?:{_SEP}{_RANGE})*)", re.IGNORECASE)
SHORT_CITATION_RE = re.compile(r"\[L(\d+)\s+S(\d+)\]")
NOT_FOUND_TAG = " [slide not found]"
ANSWER_KEY_RE = re.compile(r"^(?:#{1,4}\s*|\*\*)Answer key(?:\*\*)?:?\s*$", re.IGNORECASE | re.MULTILINE)
TOKEN_RE = re.compile(r"[a-z0-9]+")
SOURCE_MARK = "\n\n---\n**Source:** " # separates an answer from the line that names where it came from
STORE = SupabaseStore()
# --------------------------------------------------------------------------- #
# Lecture loading and indexing
# --------------------------------------------------------------------------- #
@dataclass
class Slide:
lecture: int
number: int # 1-based slide (PDF page) number
text: str
@property
def key(self) -> tuple[int, int]:
return (self.lecture, self.number)
@property
def image_only(self) -> bool:
return not self.text
@property
def section(self) -> bool:
"""A short non-topic title such as Summary or Questions."""
return 0 < len(self.text) < DIAGRAM_MAX_CHARS and SECTION_TITLE_RE.match(self.text) is not None
@property
def diagram(self) -> bool:
"""A slide that carries only a short topic title: the content is a figure."""
return 0 < len(self.text) < DIAGRAM_MAX_CHARS and not self.section
@property
def index_text(self) -> str:
if self.image_only:
return "(image-only slide)"
if self.section:
return f"{self.text} (section slide)"
if self.diagram:
return f"{self.text} (diagram slide)"
return self.text
@dataclass
class Lecture:
number: int
title: str
path: Path
slides: list[Slide]
@property
def label(self) -> str:
return f"Lecture {self.number}: {self.title}"
def _normalize(text: str) -> str:
return " ".join(text.split())
def load_lectures(folder: Path) -> dict[int, Lecture]:
lectures: dict[int, Lecture] = {}
unnumbered: list[Path] = []
for pdf in sorted(folder.glob("*.pdf")):
if pdf.name.startswith("._"): # macOS resource-fork copies inside zips
continue
match = FILENAME_RE.match(pdf.name)
if not match:
unnumbered.append(pdf)
continue
lectures[int(match.group(1))] = _read_pdf(int(match.group(1)), match.group(2).strip(), pdf)
for pdf in unnumbered: # any PDF that does not follow the naming pattern
number = max(lectures, default=0) + 1
lectures[number] = _read_pdf(number, pdf.stem, pdf)
return dict(sorted(lectures.items()))
def _read_pdf(number: int, title: str, path: Path) -> Lecture:
slides: list[Slide] = []
with fitz.open(path) as doc:
for index, page in enumerate(doc, start=1):
slides.append(Slide(number, index, _normalize(page.get_text())))
log.info("Loaded Lecture %s (%s): %d slides", number, title, len(slides))
return Lecture(number, title, path, slides)
# Question and function words. Without this list "What are the ligaments of the knee?" matched a brachial plexus
# slide that asks "What type of fibers are present? What are the consequences...?" on "what" and "are" alone.
STOP_WORDS = frozenset("""
a about an and any are as at be been but by can could describe did do does each explain for from give had has
have how i if in into is it its list me name of on or please should show so tell than that the their them then
there these they this those to us was we were what when where which while who whom whose why will with would you
your between compare define identify happen happens
""".split())
def tokenize(text: str) -> list[str]:
tokens = []
for tok in TOKEN_RE.findall(text.lower()):
if tok in STOP_WORDS:
continue
if len(tok) > 3 and tok.endswith("s"):
tok = tok[:-1]
tokens.append(tok)
return tokens
class BM25:
"""Small BM25 index."""
def __init__(self, documents: list[list[str]], k1: float = 1.5, b: float = 0.75):
self.k1, self.b = k1, b
self.doc_len = [len(d) for d in documents]
self.avg_len = (sum(self.doc_len) / len(documents)) if documents else 1.0
self.postings: dict[str, list[tuple[int, int]]] = {}
for idx, doc in enumerate(documents):
for term, freq in Counter(doc).items():
self.postings.setdefault(term, []).append((idx, freq))
n = len(documents)
self.idf = {t: math.log(1 + (n - len(p) + 0.5) / (len(p) + 0.5)) for t, p in self.postings.items()}
def scores(self, query: list[str]) -> list[float]:
out = [0.0] * len(self.doc_len)
for term in set(query):
idf = self.idf.get(term)
if idf is None:
continue
for idx, freq in self.postings[term]:
denom = freq + self.k1 * (1 - self.b + self.b * self.doc_len[idx] / self.avg_len)
out[idx] += idf * freq * (self.k1 + 1) / denom
return out
LECTURES: dict[int, Lecture] = load_lectures(LECTURES_DIR) if LECTURES_DIR.is_dir() else {}
ALL_SLIDES: list[Slide] = [s for lec in LECTURES.values() for s in lec.slides]
SLIDE_INDEX = BM25([tokenize(s.text) for s in ALL_SLIDES])
TITLE_INDEX = BM25([tokenize(lec.title) for lec in LECTURES.values()])
if not LECTURES:
log.error("No lecture PDFs found in %s", LECTURES_DIR)
def get_slide(lecture: int, number: int) -> Slide | None:
lec = LECTURES.get(lecture)
if lec and 1 <= number <= len(lec.slides):
return lec.slides[number - 1]
return None
def retrieve_slides(query: str, limit: int = MAX_CONTEXT_SLIDES) -> list[Slide]:
"""Slides whose text matches the query best, plus figure slides that follow them.
Slides come only from the lectures that match the query well (the sum of a lecture's three best slide scores
is at least LECTURE_FOCUS of the best lecture's), so a word such as "joint" or "nerve" on one slide of the
shoulder lecture does not add that slide to a question about the knee.
"""
tokens = tokenize(query)
text_scores = SLIDE_INDEX.scores(tokens)
title_scores = dict(zip(LECTURES, TITLE_INDEX.scores(tokens)))
def score(i: int) -> float:
return text_scores[i] + TITLE_WEIGHT * title_scores[ALL_SLIDES[i].lecture]
ranked = sorted((i for i, s in enumerate(text_scores) if s > 0 and not ALL_SLIDES[i].section), key=lambda i: -score(i))
by_lecture: dict[int, list[float]] = {}
for i in ranked:
by_lecture.setdefault(ALL_SLIDES[i].lecture, []).append(score(i))
strength = {lecture: sum(scores[:3]) for lecture, scores in by_lecture.items()}
best = max(strength.values(), default=0.0)
ranked = [i for i in ranked if strength[ALL_SLIDES[i].lecture] >= LECTURE_FOCUS * best]
chosen: list[Slide] = []
seen: set[tuple[int, int]] = set()
for idx in ranked[:limit]:
slide = ALL_SLIDES[idx]
for cand in (slide, get_slide(slide.lecture, slide.number + 1)):
if cand is None or cand.key in seen:
continue
if cand is slide or cand.diagram or cand.image_only:
seen.add(cand.key)
chosen.append(cand)
return chosen[:limit]
FOLLOW_UP_RE = re.compile(r"\b(?:it|its|they|them|their)\b", re.IGNORECASE)
def slide_query(mode: str, text: str, state: dict) -> str:
"""The text that picks the slide images for a request.
Quick actions (quiz, clinical, simplify, show) act on the topic of the last answer, so the structures that answer
named are added. A short follow-up question that refers back ("How is it injured?") gets them too; otherwise it
matched slides on its own words alone (injury slides of the shoulder after a question about the knee).
"""
topics = state.get("search_topics") or []
if not topics:
return text
if mode != "ask":
return " ".join([text] + topics)
if FOLLOW_UP_RE.search(text) and len(tokenize(text)) <= 3:
return " ".join([text] + topics)
return text
def build_slide_index_text() -> str:
lines = []
for lec in LECTURES.values():
lines.append(f"\n## Lecture {lec.number}: {lec.title} ({len(lec.slides)} slides)")
for s in lec.slides:
lines.append(f"[L{lec.number} S{s.number}] {s.index_text}")
return "\n".join(lines)
# --------------------------------------------------------------------------- #
# Slide rendering
# --------------------------------------------------------------------------- #
_render_lock = threading.Lock()
_render_cache: OrderedDict[tuple[int, int], bytes] = OrderedDict()
_viewer_dir = Path(tempfile.mkdtemp(prefix="anatomy-slides-"))
atexit.register(shutil.rmtree, _viewer_dir, ignore_errors=True)
def _render_page(lecture: int, number: int, width: int) -> Image.Image:
with fitz.open(LECTURES[lecture].path) as doc:
page = doc[number - 1]
zoom = width / page.rect.width
pix = page.get_pixmap(matrix=fitz.Matrix(zoom, zoom), alpha=False)
return Image.frombytes("RGB", (pix.width, pix.height), pix.samples)
def render_slide(lecture: int, number: int) -> bytes:
"""Render one PDF page to JPEG bytes for Gemini (cached)."""
key = (lecture, number)
with _render_lock:
if key in _render_cache:
_render_cache.move_to_end(key)
return _render_cache[key]
buffer = io.BytesIO()
_render_page(lecture, number, RENDER_WIDTH).save(buffer, format="JPEG", quality=85)
data = buffer.getvalue()
_render_cache[key] = data
while len(_render_cache) > RENDER_CACHE_SIZE:
_render_cache.popitem(last=False)
return data
def slide_image(lecture: int, number: int) -> str:
"""Path of a high-resolution WebP render of one slide for the viewer, written once per slide.
Gradio serves the file as-is, so the slide is not re-encoded at a lower quality on the way out.
"""
path = _viewer_dir / f"lecture{lecture}-slide{number}.webp"
with _render_lock:
if not path.is_file():
buffer = io.BytesIO()
_render_page(lecture, number, VIEWER_WIDTH).save(buffer, format="WEBP", quality=VIEWER_QUALITY)
path.write_bytes(buffer.getvalue())
return str(path)
# --------------------------------------------------------------------------- #
# Gemini
# --------------------------------------------------------------------------- #
class SlideRef(BaseModel):
lecture: int
slide: int
class TutorReply(BaseModel):
answer: str
slides: list[SlideRef]
covered_by_slides: bool
search_topics: list[str] = [] # the structures the request is about; used to look up TeachMeAnatomy and Kenhub articles
class WebReply(BaseModel):
answer: str
covered: bool
used_urls: list[str] = []
class TutorError(Exception):
"""An error with a message that can be shown to the student."""
SYSTEM_PROMPT = f"""You are the tutor of the Creighton Anatomy Learning Platform. Your only source of knowledge is the SLIDE INDEX below (text of every slide in the course lecture PDFs) and the slide images attached to a request.
Rules:
1. Answer only from the slides. Every factual statement must be supported by a slide and cited inline in exactly this form: (Lecture N, slide M). Cite the specific slide that states the fact. Never use the [L.. S..] index labels in your answer.
2. If the slides do not cover the request, say so in one plain sentence. Never answer from outside knowledge and never add a section with material that is not on the slides: when the slides are not enough, the platform looks the topic up on TeachMeAnatomy and Kenhub separately.
3. Never invent structures, relationships, mnemonics, numbers, clinical facts, or slide numbers. Only cite slide numbers that exist in the SLIDE INDEX.
4. Write for a health-professions student: accurate, well organized, concise. Use Markdown headings, bullet lists, and bold for key terms where helpful. Do not use emoji.
5. Attached slide images were found by keyword search and may be irrelevant; use them only when they match the question. Slides marked "(diagram slide)" carry only a short title and usually show a figure on that topic; slides marked "(image-only slide)" have no text and usually continue the topic of the preceding slide; slides marked "(section slide)" are summaries, reviews, or question slides.
6. In the "slides" field list only the slides whose images directly illustrate the answer, most useful first, at most {MAX_VIEWER_SLIDES}. Do not pad the list: one to three slides is typical, and it must be empty when no slide illustrates the answer. Every slide listed must be cited in the answer, or be a diagram or image-only slide that directly follows a cited slide.
7. Set "covered_by_slides" to false when the slides do not address the request.
8. In "search_topics" list the anatomical structures or regions the request is about, one entry per structure, at most three, each one to four words naming the structure itself in standard anatomical terms, for example ["knee joint", "ankle joint"] or ["brachial plexus"]. For a small part of a joint, organ, or region, add that joint, organ, or region as another entry, for example ["anterior cruciate ligament", "knee joint"]. Use the common names that articles are filed under, for example "ankle joint" rather than "talocrural joint" and "shoulder joint" rather than "glenohumeral joint". Do not add words such as injury, fracture, clinical, test, or function, and never list a structure the request is not about. These entries are used to look up TeachMeAnatomy and Kenhub articles.
9. Answer the current request. Use earlier turns only to understand what it refers to ("it", "what about the knee?"); do not include material about an earlier topic unless the request asks for a comparison.
SLIDE INDEX
Each line is one slide: [L<lecture> S<slide>] followed by the slide text.
{build_slide_index_text()}
"""
WEB_SYSTEM_PROMPT = """You are the tutor of the Creighton Anatomy Learning Platform. Your only source of knowledge is the REFERENCE PAGES attached to the request: article excerpts from TeachMeAnatomy (teachmeanatomy.info) and Kenhub (kenhub.com).
Rules:
1. Use only the reference pages. Every factual statement must be supported by a page and cited inline in exactly this form: (TeachMeAnatomy: page title) or (Kenhub: page title), using the page titles given in the REFERENCE PAGES headers.
2. Never add information from your own knowledge and never add a section for material that is not on the pages. If the pages do not answer the request, say so in one plain sentence and set "covered" to false.
3. Use only pages about the structures the request is about; ignore any page about a different structure or region.
4. Never invent structures, relationships, mnemonics, numbers, or clinical facts. Explain in your own words; do not copy long passages.
5. Write for a health-professions student: accurate, well organized, concise. Use Markdown headings, bullet lists, and bold for key terms where helpful. Headings name the content, for example "### Ligaments of the knee"; never write headings about sources such as "From the reference pages" or "Beyond the reference pages", and do not mention the lecture slides. Do not use emoji.
6. In "used_urls" list the URLs, from the REFERENCE PAGES headers, of the pages you actually used, most useful first. Leave it empty when "covered" is false.
7. Answer the current request. Use earlier turns only to understand what it refers to; do not include material about an earlier topic unless the request asks for a comparison.
"""
TASK_PROMPTS = {
"ask": "Student question: {question}",
"quiz": (
"Create a five-question multiple-choice quiz on this topic: {topic}\n"
"Use only facts stated on the slides. Number the questions, give options A to D, and make exactly "
"one option correct. After the questions add a section headed '### Answer key' that gives, for each "
"question, the correct letter, a one-sentence explanation, and the slide citation."
),
"clinical": (
"List the clinical correlations for this topic that the slides state: {topic}\n"
"Give each clinical point with its slide citation, as a bullet list, without a heading. Include only clinical "
"points that are stated on the slides; do not add any point from outside the slides. If the slides state "
"no clinical information on this topic, write exactly: The lecture slides do not include clinical notes "
"on this topic. and set \"covered_by_slides\" to false."
),
"simplify": (
"Rewrite your previous answer in plain language for a student meeting this material for the first time. "
"Use short sentences, define any technical term the first time it appears, keep every citation and section "
"heading, and do not add facts that were not in the previous answer.\n\n"
"Previous answer:\n{previous}"
),
"show": (
"Select only the slides whose images directly illustrate this topic, most useful first and at most "
"{max_slides}; do not pad the list. Topic: {topic}\n"
"Put them in the \"slides\" field. In \"answer\" write one or two sentences that cite every chosen slide as "
"(Lecture N, slide M) and say what each shows, or exactly: No slide in the lecture files illustrates this topic."
),
}
# The same requests, answered from TeachMeAnatomy and Kenhub pages (see WEB_SYSTEM_PROMPT).
WEB_TASK_PROMPTS = {
"ask": "Student question: {question}",
"quiz": (
"Create a five-question multiple-choice quiz on this topic: {topic}\n"
"Use only facts stated on the reference pages. Number the questions, give options A to D, and make exactly "
"one option correct. After the questions add a section headed '### Answer key' that gives, for each "
"question, the correct letter, a one-sentence explanation, and the page citation."
),
"clinical": (
"List the clinical correlations for this topic that the reference pages state: {topic}\n"
"Give each clinical point with its page citation, as a bullet list, without a heading. Include only "
"clinical points that are stated on the pages. If the pages state no clinical information on this topic, "
"write exactly: The pages checked do not include clinical notes on this topic. and set \"covered\" to false."
),
}
_client = (
genai.Client(
api_key=GEMINI_API_KEY,
http_options=genai_types.HttpOptions(
timeout=GEMINI_TIMEOUT_MS,
retry_options=genai_types.HttpRetryOptions(attempts=3, initial_delay=1.0, max_delay=8.0),
),
)
if GEMINI_API_KEY
else None
)
_active_model = GEMINI_MODEL
_model_lock = threading.Lock()
def _candidate_models() -> list[str]:
models = [_active_model]
models += [m for m in [GEMINI_MODEL] + FALLBACK_MODELS if m not in models]
return models
def _student_message(err: genai_errors.APIError) -> str:
code = err.code or 0
if code == 429:
return "The tutor is busy right now. Please try again in a minute."
if code in (401, 403):
return "The Gemini API key was not accepted. Ask the course administrator to check the GEMINI_API_KEY secret."
if code >= 500:
return "The Gemini service is temporarily unavailable. Please try again."
return "The Gemini API rejected the request. Please try again or rephrase the question."
def _history_contents(history: list[dict]) -> list[genai_types.Content]:
return [
genai_types.Content(role=turn["role"], parts=[genai_types.Part.from_text(text=turn["text"])])
for turn in history[-2 * MAX_HISTORY_TURNS:]
]
def _generate(contents: list[genai_types.Content], system_prompt: str, schema: type[BaseModel]) -> str:
"""Send one request to Gemini and return the JSON text of the reply."""
global _active_model
if _client is None:
raise TutorError(
"The Gemini API key is not configured. Add GEMINI_API_KEY as a secret in the Space settings "
"and restart the Space."
)
config = genai_types.GenerateContentConfig(
system_instruction=system_prompt,
temperature=0.2,
response_mime_type="application/json",
response_schema=schema,
)
response = None
for model in _candidate_models():
try:
response = _client.models.generate_content(model=model, contents=contents, config=config)
except genai_errors.ClientError as err:
if err.code == 404:
log.warning("Model %s not available (%s); trying the next model", model, err.message)
continue
log.error("Gemini client error %s: %s", err.code, err.message)
raise TutorError(_student_message(err)) from err
except genai_errors.APIError as err:
log.error("Gemini API error %s: %s", err.code, err.message)
raise TutorError(_student_message(err)) from err
except Exception as err: # transport failures, timeouts, malformed bodies
log.exception("Gemini request failed")
raise TutorError("The tutor could not reach the Gemini service. Please try again.") from err
with _model_lock:
if _active_model != model:
log.info("Using Gemini model %s", model)
_active_model = model
break
if response is None:
raise TutorError("None of the configured Gemini models are available for this API key.")
usage = getattr(response, "usage_metadata", None)
if usage is not None:
log.info(
"Gemini %s: prompt=%s cached=%s output=%s tokens",
_active_model, usage.prompt_token_count, usage.cached_content_token_count, usage.candidates_token_count,
)
feedback = getattr(response, "prompt_feedback", None)
if feedback is not None and feedback.block_reason:
log.warning("Prompt blocked: %s", feedback.block_reason)
raise TutorError("The request was blocked by the Gemini content filter. Please rephrase the question.")
candidates = getattr(response, "candidates", None) or []
finish = getattr(candidates[0], "finish_reason", None) if candidates else None
finish_name = getattr(finish, "name", str(finish)) if finish is not None else "STOP"
if finish_name not in ("STOP", "FINISH_REASON_UNSPECIFIED"):
log.warning("Gemini finished with %s", finish_name)
if finish_name == "MAX_TOKENS":
raise TutorError("The answer was too long to complete. Please ask a narrower question.")
raise TutorError("Gemini stopped before completing the answer. Please try again or rephrase the question.")
text = response.text
if not text:
raise TutorError("Gemini returned an empty response. Please try again or rephrase the question.")
return text
def gemini_reply(task: str, history: list[dict], context_slides: list[Slide]) -> TutorReply:
"""Answer a request from the slides and return the parsed structured reply."""
if not LECTURES:
raise TutorError(f"No lecture PDFs were found in {LECTURES_DIR}. Upload the Lectures folder next to app.py.")
contents = _history_contents(history)
parts = []
for slide in context_slides:
parts.append(genai_types.Part.from_text(text=f"Attached image: Lecture {slide.lecture}, slide {slide.number}"))
parts.append(genai_types.Part.from_bytes(data=render_slide(slide.lecture, slide.number), mime_type="image/jpeg"))
parts.append(genai_types.Part.from_text(text=task))
contents.append(genai_types.Content(role="user", parts=parts))
text = _generate(contents, SYSTEM_PROMPT, TutorReply)
try:
reply = TutorReply.model_validate_json(text)
except ValidationError as err:
log.error("Could not parse Gemini reply: %s", text[:500])
raise TutorError("Gemini returned a response in an unexpected format. Please try again.") from err
reply.slides = _valid_slides(reply.slides)
reply.answer = _drop_beyond_sections(reply.answer.strip())
reply.search_topics = [t.strip() for t in reply.search_topics if t and t.strip()][:web_sources.MAX_TOPICS]
return reply
def gemini_web_reply(task: str, history: list[dict], pages: list[web_sources.Page]) -> WebReply:
"""Answer a request from reference pages and return the parsed structured reply."""
excerpts = "\n\n".join(f"=== Page {i}: {page.label}\nURL: {page.url}\n{page.text}" for i, page in enumerate(pages, 1))
contents = _history_contents(history)
contents.append(genai_types.Content(role="user", parts=[
genai_types.Part.from_text(text=f"REFERENCE PAGES\n{excerpts}"),
genai_types.Part.from_text(text=task),
]))
text = _generate(contents, WEB_SYSTEM_PROMPT, WebReply)
try:
reply = WebReply.model_validate_json(text)
except ValidationError as err:
log.error("Could not parse Gemini web reply: %s", text[:500])
raise TutorError("Gemini returned a response in an unexpected format. Please try again.") from err
known = {page.url.rstrip("/") for page in pages}
reply.used_urls = [url for url in dict.fromkeys(u.strip().rstrip("/") for u in reply.used_urls) if url in known]
reply.answer = _drop_beyond_sections(reply.answer.strip())
return reply
REWRITE_SYSTEM_PROMPT = """You are the tutor of the Creighton Anatomy Learning Platform. The request contains a previous answer that cites TeachMeAnatomy or Kenhub pages, and possibly lecture slides. Rewrite it as asked, keeping every citation and every section heading exactly as written, and add no facts that are not in the previous answer. Do not use emoji. Leave the "slides" field empty, set "covered_by_slides" to false, and leave "search_topics" empty.
"""
def gemini_rewrite(task: str, history: list[dict]) -> TutorReply:
"""Rewrite a previous web-sourced answer (Simplify) without the slide index, so its sources stay the same."""
contents = _history_contents(history)
contents.append(genai_types.Content(role="user", parts=[genai_types.Part.from_text(text=task)]))
text = _generate(contents, REWRITE_SYSTEM_PROMPT, TutorReply)
try:
reply = TutorReply.model_validate_json(text)
except ValidationError as err:
log.error("Could not parse Gemini rewrite: %s", text[:500])
raise TutorError("Gemini returned a response in an unexpected format. Please try again.") from err
reply.slides = []
reply.answer = _drop_beyond_sections(reply.answer.strip())
return reply
# A heading, or a line standing alone as a heading (bold, italic, or ending in a colon), that announces material
# beyond the sources: "### Beyond the slides", "### **Beyond the reference pages**", "### 2. Beyond the slides",
# "**Beyond the slides**", "Beyond the slides:", "### Additional points (beyond the slides)".
BEYOND_RE = re.compile(r"\bbeyond\s+(?:the\s+)?(?:lecture\s+)?(?:slides|reference\s+pages|pages|sources|lectures?)\b", re.IGNORECASE)
HEADING_LEVEL_RE = re.compile(r"^\s*(#{1,6})\s")
PSEUDO_HEADING_RE = re.compile(r"^\s*(?:\*\*[^*].*\*\*|\*[^*].*\*|__.+__|[^\s#*\-][^:]{0,80}:)\s*$")
def _drop_beyond_sections(markdown: str) -> str:
"""Remove any "Beyond the slides" / "Beyond the reference pages" section from an answer.
The prompts forbid such sections; this makes sure a model that writes one anyway never shows students
material that does not come from the slides or the reference pages. A section announced by a Markdown heading
ends at the next heading of the same or a higher level; one announced by a stand-alone line ends at the next
heading of level 1 to 3 or the next stand-alone line.
"""
out: list[str] = []
skipping = 0 # 0: keeping; 1-6: inside a heading section; 7: inside a pseudo heading
for line in markdown.split("\n"):
level_match = HEADING_LEVEL_RE.match(line)
level = len(level_match.group(1)) if level_match else 0
pseudo = not level and bool(PSEUDO_HEADING_RE.match(line))
if skipping:
if (skipping <= 6 and level and level <= skipping) or (skipping == 7 and ((level and level <= 3) or pseudo)):
skipping = 0
else:
continue
if (level or pseudo) and BEYOND_RE.search(line):
skipping = level or 7
log.warning("Removed a section that is not grounded in the sources: %s", line.strip()[:80])
continue
out.append(line)
return "\n".join(out).strip()
def _valid_slides(refs: list[SlideRef]) -> list[SlideRef]:
seen: set[tuple[int, int]] = set()
valid = []
for ref in refs:
if get_slide(ref.lecture, ref.slide) is not None and (ref.lecture, ref.slide) not in seen:
seen.add((ref.lecture, ref.slide))
valid.append(ref)
return valid[:MAX_VIEWER_SLIDES]
def _verify_citations(answer: str) -> str:
"""Flag citations that point to slides that do not exist."""
answer = SHORT_CITATION_RE.sub(r"(Lecture \1, slide \2)", answer.replace(NOT_FOUND_TAG, ""))
def check(match: re.Match) -> str:
lecture = int(match.group(1))
for part in re.split(_SEP, match.group(2), flags=re.IGNORECASE):
bounds = [int(n) for n in re.findall(r"\d+", part)]
if not bounds:
continue
first, last = bounds[0], bounds[-1]
if first > last or any(get_slide(lecture, n) is None for n in range(first, last + 1)):
return match.group(0) + NOT_FOUND_TAG
return match.group(0)
return CITATION_RE.sub(check, answer)
def _cited_slides(answer: str) -> list[tuple[int, int]]:
"""Existing slides cited in an answer as (Lecture N, slide M), in order of first mention."""
found: dict[tuple[int, int], None] = {}
for match in CITATION_RE.finditer(answer):
lecture = int(match.group(1))
for part in re.split(_SEP, match.group(2), flags=re.IGNORECASE):
bounds = [int(n) for n in re.findall(r"\d+", part)]
if not bounds:
continue
first, last = bounds[0], bounds[-1]
if first > last:
continue
for number in range(first, last + 1):
if get_slide(lecture, number) is not None:
found[(lecture, number)] = None
return list(found)
def _figure_anchor(lecture: int, number: int) -> tuple[int, int]:
"""The nearest preceding slide with text: figure slides continue the topic of that slide."""
while number > 1:
slide = get_slide(lecture, number)
if slide is None or not (slide.diagram or slide.image_only):
break
number -= 1
return (lecture, number)
def _relevant_slides(refs: list[SlideRef], answer: str) -> list[SlideRef]:
"""Only the chosen slides the answer actually rests on: cited slides and the figure slides that follow them.
When none of the chosen slides is cited, the cited slides themselves are shown instead.
"""
cited = _cited_slides(answer)
cited_set = set(cited)
kept = [ref for ref in refs if (ref.lecture, ref.slide) in cited_set or _figure_anchor(ref.lecture, ref.slide) in cited_set]
return kept or [SlideRef(lecture=lecture, slide=number) for lecture, number in cited[:MAX_VIEWER_SLIDES]]
def _collapse_answer_key(markdown: str) -> str:
"""Hide a quiz answer key behind a collapsible block."""
match = ANSWER_KEY_RE.search(markdown)
if not match:
return markdown
questions, key = markdown[: match.start()].rstrip(), markdown[match.end():].strip()
return f"{questions}\n\n<details><summary>Show answer key</summary>\n\n{key}\n\n</details>"
# --------------------------------------------------------------------------- #
# Source notes: every answer ends with the line that names where it came from
# --------------------------------------------------------------------------- #
def _slides_note(answer: str) -> str:
cited = _cited_slides(answer)
if not cited:
return "Course lecture slides."
by_lecture: dict[int, list[int]] = {}
for lecture, number in cited:
by_lecture.setdefault(lecture, []).append(number)
parts = []
for lecture, numbers in by_lecture.items():
numbers = sorted(set(numbers))
label = "slides" if len(numbers) > 1 else "slide"
parts.append(f"Lecture {lecture} ({LECTURES[lecture].title}), {label} {', '.join(map(str, numbers))}")
return "Course lecture slides: " + "; ".join(parts) + "."
def _pages_list(pages: list[web_sources.Page]) -> str:
return "; ".join(f"{page.site}: [{page.title}]({page.url})" for page in pages)
def _web_note(pages: list[web_sources.Page]) -> str:
return "The lecture slides do not cover this request. Answered from " + _pages_list(pages) + "."
def _site_names(pages: list[web_sources.Page]) -> str:
""" "TeachMeAnatomy", "Kenhub", or "TeachMeAnatomy and Kenhub", in that order, for the pages used."""
names = [site.name for site in web_sources.SITES if any(page.site == site.name for page in pages)]
return " and ".join(names) or "TeachMeAnatomy and Kenhub"
def _demote_headings(markdown: str) -> str:
"""Headings inside a section become level-4 headings, below the section's own level-3 heading."""
return re.sub(r"(?m)^\s*#{1,4}\s+", "#### ", markdown)
def _none_note(pages: list[web_sources.Page], problem: str = "") -> str:
if pages:
note = "Not found in the lecture slides. The reference pages checked (" + _pages_list(pages) + ") do not answer it either."
else:
note = "Not found in the lecture slides, and no matching article was found on TeachMeAnatomy or Kenhub."
return f"{note} ({problem}.)" if problem else note
def _with_source(content: str, note: str) -> str:
return content + SOURCE_MARK + note if note else content
def _split_source(content: str) -> tuple[str, str]:
"""(answer, source note) of a stored message."""
body, _, note = content.partition(SOURCE_MARK)
return body, note
# --------------------------------------------------------------------------- #
# Sessions (server-side, one per browser tab) and viewer helpers
# --------------------------------------------------------------------------- #
def new_state() -> dict:
return {
"history": [], "topic": "", "last_answer": "", "last_note": "", "last_source": "", "slides": [], "slide_idx": 0,
"view": None, "mode": "lecture", "no_slide_topic": "",
"search_topics": [], # structures the current topic is about, as named by the last answer (for Clinical Correlation)
}
class Session:
"""Chat transcript, signed-in user, and viewer state for one browser session, guarded by a lock."""
def __init__(self) -> None:
self.lock = threading.Lock()
self.resume_lock = threading.Lock() # one automatic sign-in at a time (see _resume)
self.busy = False
self.user_id: str | None = None
self.email: str | None = None
self.session_id: str | None = None # one per visit; the key of the anatomy_sessions row
self.ended = False # the visit's row was closed because the tab lost its connection
self.epoch = 0 # bumped by Sign in, Create account, and Sign out
self.signed_out = False # Sign out was pressed: never sign this tab back in automatically
self.saved_at = 0.0 # when this tab last queued an answer for Supabase
self.awaiting = False # an old tab of this student was still writing an answer at sign-in
self.last_beat = 0.0 # when time on the site was last written to Supabase
# Bumped whenever the chat changes; the page keeps the value it last received, so a page that missed an
# answer (its connection dropped while the answer was being written) can be detected and brought up to
# date. Starts from the clock so that numbers from before a server restart never match.
self.version = int(time.time() * 1000)
self.messages: list[dict] = []
self.state = new_state()
SESSIONS: OrderedDict[str, Session] = OrderedDict()
_sessions_lock = threading.Lock()
def get_session(request) -> Session:
key = getattr(request, "session_hash", None) or "default"
evicted: list[Session] = []
with _sessions_lock:
session = SESSIONS.get(key)
if session is None:
while len(SESSIONS) >= MAX_SESSIONS:
# Drop the least recently used session, anonymous ones first, so that a flood of
# anonymous page loads cannot push signed-in students out of memory.
victim = next((k for k, s in SESSIONS.items() if s.user_id is None), next(iter(SESSIONS)))
evicted.append(SESSIONS.pop(victim))
session = SESSIONS[key] = Session()
else:
SESSIONS.move_to_end(key)
for old in evicted:
if old.session_id and not old.ended:
STORE.enqueue("end_session", old.session_id)
return session
def forget_session(request: gr.Request) -> None:
"""The tab closed or its connection dropped: close the stored visit. The in-memory session is kept so that a
tab that merely reconnects keeps working (its next activity opens a new visit, see _reopen); memory stays
bounded by MAX_SESSIONS."""
key = getattr(request, "session_hash", None)
with _sessions_lock:
session = SESSIONS.get(key)
if session is None:
return
with session.lock:
session_id = session.session_id if not session.ended else None
if session_id:
session.ended = True
if session_id:
STORE.enqueue("end_session", session_id)
def _reopen(session: Session) -> None:
"""A tab that comes back after its visit was closed (laptop asleep, connection lost) starts a new visit, so the
time away is not counted as time on the site."""
with session.lock:
if not (session.user_id and session.ended):
return
session.session_id, session.ended, session.last_beat = uuid.uuid4().hex, False, time.monotonic()
session_id, user_id = session.session_id, session.user_id
STORE.enqueue("start_session", session_id, user_id)
def _bump(session: Session) -> int:
"""Record that the chat changed (call with session.lock held). Returns the new version."""
session.version += 1
return session.version
def _caption(state: dict) -> str:
view = state.get("view")
if not view:
return "Ask a question, or choose a lecture below to browse its slides."
lecture, number = view
lec = LECTURES[lecture]
text = f"**Lecture {lecture}: {lec.title}** \nSlide {number} of {len(lec.slides)}"
slides = state.get("slides") or []
if state.get("mode") == "related" and view in slides:
text += f" \nRelated slide {slides.index(view) + 1} of {len(slides)}"
return text
def viewer_outputs(state: dict, caption: str | None = None) -> tuple:
"""Values for (slide image, caption, lecture dropdown, slide number)."""
view = state.get("view")
if not view:
return None, caption or _caption(state), gr.update(value=None), gr.update(value=1)
lecture, number = view
return slide_image(lecture, number), caption or _caption(state), gr.update(value=lecture), gr.update(value=number)
def set_view(state: dict, lecture: int, number: int, mode: str) -> dict:
state["view"] = [lecture, number]
state["mode"] = mode
return state
def controls(state: dict, enabled: bool) -> tuple:
"""Interactivity for (Show Image, Quiz Me, Clinical Correlation, Simplify, Ask, Clear)."""
quick = gr.update(interactive=enabled and bool(state.get("topic")))
return (quick,) * 4 + (gr.update(interactive=enabled),) * 2
KEEP = gr.update()
KEEP_VIEWER = (KEEP,) * 4
NOOP_TURN = (KEEP,) * 13 # chat, question, viewer (4), controls (6), version
NOOP_SHOW = (KEEP,) * 12 # chat, viewer (4), controls (6), version
NOOP_LOGIN = (KEEP,) * 20 # login panel, app panel, message, badge, timer, chat, question, viewer (4), controls (6), user ID, email, version
NOOP_SYNC = (KEEP,) * 14 # chat, question, version, badge, viewer (4), controls (6)
def _pending(text: str) -> dict:
"""A chat message that shows a spinner while the tutor works."""
return {"role": "assistant", "content": text, "metadata": {"title": "Thinking", "status": "pending"}}
# Appended to clinical source notes by the previous version; removed when such a note is reused.
OLD_TUTOR_NOTE = " The 'Beyond the slides' section is standard clinical anatomy from the tutor, not from the slides."
SIGNED_OUT_TEXT = ("Your sign-in could not be restored after the site restarted. "
"Please reload the page and sign in again; your chat is saved.")
def _track(session: Session, event: str, detail: dict | None = None) -> None:
"""Record a button click for the signed-in user (call without holding the session lock)."""
_reopen(session)
if session.user_id and session.session_id:
STORE.enqueue("log_event", session.user_id, session.session_id, event, detail or {})
def _save(session: Session, rows: list[dict]) -> None:
"""Store chat messages for the signed-in user."""
if session.user_id and session.session_id and rows:
STORE.enqueue("append_messages", session.user_id, session.session_id, rows)
# --------------------------------------------------------------------------- #
# Answering: slides first, then the reference sites
# --------------------------------------------------------------------------- #
@dataclass
class Outcome:
text: str # what the student sees, before the source note
model_text: str # what enters the Gemini conversation history
slides: list[list[int]] # [lecture, slide] pairs for the viewer
source: str # slides, web, slides+web, none, error
note: str # the source note
topics: list[str] = field(default_factory=list) # structures the request is about (reply.search_topics)
def _answer(mode: str, task: str, topic: str, history: list[dict], state: dict):
"""Generator: yields progress text for the chat, returns the Outcome."""
if mode == "simplify": # a rewrite of the previous answer keeps that answer's sources
source = state.get("last_source") or "slides"
if source == "slides":
yield "Rewriting the previous answer from the lecture slides..."
reply = gemini_reply(task, history, retrieve_slides(slide_query(mode, topic, state)))
answer = _verify_citations(reply.answer)
slides = _relevant_slides(reply.slides, answer)
else:
yield "Rewriting the previous answer..."
reply = gemini_rewrite(task, history)
answer = _verify_citations(reply.answer) # a clinical answer can also cite slides
slides = _relevant_slides([], answer)
return Outcome(answer, reply.answer, [[s.lecture, s.slide] for s in slides], source, state.get("last_note", ""))
if mode == "clinical":
return (yield from _clinical(task, topic, history, state.get("search_topics") or [],
slide_query(mode, topic, state)))
yield "Searching the lecture slides..."
reply = gemini_reply(task, history, retrieve_slides(slide_query(mode, topic, state)))
topics = reply.search_topics
answer = _verify_citations(reply.answer)
if reply.covered_by_slides:
slides = _relevant_slides(reply.slides, answer)
return Outcome(answer, reply.answer, [[s.lecture, s.slide] for s in slides], "slides", _slides_note(answer), topics)
if mode not in FALLBACK_MODES:
return Outcome(answer, reply.answer, [], "none", "Not found in the lecture slides.", topics)
yield "The lecture slides do not cover this request. Checking TeachMeAnatomy and Kenhub..."
pages, problem = web_sources.find_pages(topics or web_sources.question_topics(topic))
if not pages:
return Outcome(answer, reply.answer, [], "none", _none_note([], problem), topics)
web_task = WEB_TASK_PROMPTS[mode].format(question=topic, topic=topic)
web = gemini_web_reply(web_task, history, pages)
if not web.covered:
return Outcome(web.answer or answer, web.answer or reply.answer, [], "none", _none_note(pages, problem), topics)
used = [page for page in pages if page.url.rstrip("/") in web.used_urls] or pages
return Outcome(web.answer, web.answer, [], "web", _web_note(used), topics)
def _clinical_pages(topics: list[str], topic: str, history: list[dict]):
"""The clinical points stated on the TeachMeAnatomy and Kenhub pages about the topic's structures.
Returns (pages checked, note about problems, reply or None)."""
pages, problem = web_sources.find_pages(topics)
if not pages:
return pages, problem, None
try:
return pages, problem, gemini_web_reply(WEB_TASK_PROMPTS["clinical"].format(topic=topic), history, pages)
except TutorError as err: # the slide part can still be shown
log.warning("Clinical notes from the reference pages failed: %s", err)
return pages, "; ".join(filter(None, [problem, "the pages could not be read right now"])), None
def _clinical(task: str, topic: str, history: list[dict], known_topics: list[str], query: str = ""):
"""Clinical correlations: the clinical points stated on the slides, then those stated on TeachMeAnatomy and
Kenhub, each under a heading that names its source. Nothing comes from the tutor's own knowledge.
When the structures of the topic are already known from the answer that set it, the reference pages are read
while the slides are searched; otherwise after, with the structures that the slide search names."""
with ThreadPoolExecutor(max_workers=1) as pool:
pending = pool.submit(_clinical_pages, known_topics, topic, history) if known_topics else None
yield ("Searching the lecture slides, TeachMeAnatomy, and Kenhub for clinical notes..." if pending
else "Searching the lecture slides for clinical notes...")
reply = gemini_reply(task, history, retrieve_slides(query or topic))
if pending is None:
yield "Checking TeachMeAnatomy and Kenhub for clinical notes..."
pages, problem, web = _clinical_pages(reply.search_topics or web_sources.question_topics(topic), topic, history)
else:
pages, problem, web = pending.result()
slide_text = _verify_citations(reply.answer) if reply.covered_by_slides else ""
slides = _relevant_slides(reply.slides, slide_text) if slide_text else []
used = ([page for page in pages if page.url.rstrip("/") in web.used_urls] or pages) if web is not None and web.covered else []
web_text = web.answer if used else ""
topics = known_topics or reply.search_topics
if not slide_text and not web_text:
return Outcome(
"Neither the lecture slides nor the TeachMeAnatomy and Kenhub pages checked include clinical notes on this topic.",
"No clinical notes were found on this topic.", [], "none", _none_note(pages, problem), topics,
)
if web_text:
web_part = _demote_headings(web_text)
elif not pages:
web_part = "No TeachMeAnatomy or Kenhub article about this topic was found."
elif web is None:
web_part = "The TeachMeAnatomy and Kenhub pages could not be read right now."
else:
web_part = "The TeachMeAnatomy and Kenhub pages checked do not include clinical notes on this topic."
parts = [
"### From the lecture slides\n\n"
+ (_demote_headings(slide_text) if slide_text else "The lecture slides do not include clinical notes on this topic."),
f"### From {_site_names(used)}\n\n" + web_part,
]
text = "\n\n".join(parts)
if slide_text and web_text:
source = "slides+web"
note = _slides_note(slide_text).rstrip(".") + "; " + _pages_list(used) + "."
elif slide_text:
source = "slides"
if pages and web is not None:
extra = f" The pages checked ({_pages_list(pages)}) add no clinical notes."
elif pages:
extra = f" The pages found ({_pages_list(pages)}) could not be read right now."
else:
extra = " No TeachMeAnatomy or Kenhub article about this topic was found."
note = _slides_note(slide_text) + extra + (f" ({problem}.)" if problem and web is not None else "")
else:
source = "web"
note = "The lecture slides do not include clinical notes on this topic. Answered from " + _pages_list(used) + "."
return Outcome(text, text, [[s.lecture, s.slide] for s in slides], source, note, topics)
# --------------------------------------------------------------------------- #
# Event handlers
# --------------------------------------------------------------------------- #
def run_turn(mode: str, question: str, session: Session, creds: tuple = ()):
"""One tutor turn. `creds` is (user ID, email, page version) from the page, used to sign the student back in
when the server no longer knows them (after a restart)."""
question = (question or "").strip()
if session.user_id is None and creds:
_resume(session, *creds)
_track(session, mode, {"chars": len(question)} if mode == "ask" else {})
with session.lock:
st = session.state
if session.user_id is None: # the page still shows the app, but the student could not be restored
gr.Warning(SIGNED_OUT_TEXT)
yield NOOP_TURN
return
if session.busy or (mode == "ask" and not question):
yield NOOP_TURN
return
if mode == "ask":
topic, shown = question, question
task = TASK_PROMPTS["ask"].format(question=question)
elif mode == "simplify":
topic, shown = st["topic"], "Simplify the previous explanation."
# Answers saved before this version can contain a "Beyond ..." section; never rewrite it back in.
task = TASK_PROMPTS["simplify"].format(previous=_drop_beyond_sections(st["last_answer"]))
else: # quiz, clinical
topic = st["topic"]
shown = ("Quiz me on: " if mode == "quiz" else "Clinical correlation for: ") + topic
task = TASK_PROMPTS[mode].format(topic=topic)
session.busy = True
session.messages = session.messages + [{"role": "user", "content": shown}]
messages, history, snapshot = list(session.messages), list(st["history"]), dict(st)
clear = "" if mode == "ask" else gr.update()
# The answer is worked out on its own thread: Gradio stops running an event's generator as soon as the page's
# connection drops (it marks the event dead), which used to abandon an answer between two of its steps, never
# saved. The thread always finishes and saves; this generator only passes its progress on to the page.
updates: queue.Queue = queue.Queue()
threading.Thread(
target=_work_turn, args=(session, mode, task, topic, shown, history, snapshot, updates),
name="tutor-turn", daemon=True,
).start()
while True:
kind, value = updates.get()
if kind == "progress":
yield (messages + [_pending(value)], clear) + KEEP_VIEWER + controls(snapshot, False) + (KEEP,)
else:
messages, snapshot, version, failed = value
viewer = KEEP_VIEWER if failed else viewer_outputs(snapshot)
yield (messages, clear) + viewer + controls(snapshot, True) + (version,)
return
def _work_turn(session: Session, mode: str, task: str, topic: str, shown: str, history: list[dict], snapshot: dict,
updates: queue.Queue) -> None:
"""Work out one turn, record it in the session and in Supabase, and report to `updates`:
("progress", text) for each step, then ("done", (messages, state, version, failed))."""
outcome = None
try:
try:
steps = _answer(mode, task, topic, history, snapshot)
while True:
try:
updates.put(("progress", next(steps)))
except StopIteration as done:
outcome = done.value
break
except TutorError as err:
outcome = Outcome(str(err), "", [], "error", "")
except Exception:
log.exception("Unexpected error during a turn")
outcome = Outcome("Something went wrong while answering. Please try again.", "", [], "error", "")
with session.lock:
st = session.state
if outcome.source == "error":
content = outcome.text
else:
content = _collapse_answer_key(outcome.text) if mode == "quiz" else outcome.text
content = _with_source(content, outcome.note)
st["history"] = (st["history"] + [
{"role": "user", "text": task if mode == "ask" else shown},
{"role": "model", "text": outcome.model_text},
])[-2 * MAX_HISTORY_TURNS:]
st["topic"] = topic
if outcome.topics:
st["search_topics"] = outcome.topics
elif mode == "ask":
st["search_topics"] = []
if mode != "quiz":
st["last_answer"] = outcome.model_text
st["last_note"] = outcome.note
st["last_source"] = outcome.source
if outcome.slides:
st["slides"] = outcome.slides
st["slide_idx"] = 0
set_view(st, *st["slides"][0], mode="related")
elif mode == "ask":
st["slides"] = []
st["slide_idx"] = 0
st["view"] = None
session.messages = session.messages + [{"role": "assistant", "content": content}]
result = (list(session.messages), dict(st), _bump(session), outcome.source == "error")
_save(session, [
{"role": "user", "content": shown, "mode": mode, "topic": topic, "source": None},
{"role": "assistant", "content": content, "mode": mode, "topic": topic, "source": outcome.source},
])
session.saved_at = time.monotonic()
except Exception: # never leave the session busy or the page waiting
log.exception("Could not record a turn")
with session.lock:
result = (list(session.messages), dict(session.state), _bump(session), True)
finally:
with session.lock:
session.busy = False
updates.put(("done", result))
def on_ask(question, user_id, email, version, request: gr.Request):
yield from run_turn("ask", question, get_session(request), (user_id, email, version))
def on_quiz(user_id, email, version, request: gr.Request):
yield from run_turn("quiz", "", get_session(request), (user_id, email, version))
def on_clinical(user_id, email, version, request: gr.Request):
yield from run_turn("clinical", "", get_session(request), (user_id, email, version))
def on_simplify(user_id, email, version, request: gr.Request):
yield from run_turn("simplify", "", get_session(request), (user_id, email, version))
def on_show_image(user_id, email, version, request: gr.Request):
"""Show the slides for the latest answer; ask Gemini to pick some if none were chosen."""
session = get_session(request)
if session.user_id is None:
_resume(session, user_id, email, version)
_track(session, "show_image")
with session.lock:
st = session.state
if session.user_id is None:
gr.Warning(SIGNED_OUT_TEXT)
yield NOOP_SHOW
return
if session.busy or not st["topic"]:
yield NOOP_SHOW
return
needs_pick = not st["slides"] and st["topic"] != st["no_slide_topic"]
if needs_pick:
session.busy = True
topic, history, snapshot = st["topic"], list(st["history"]), dict(st)
messages = list(session.messages)
if needs_pick:
reply = None
try:
yield (messages + [_pending("Choosing slides that illustrate this topic...")],) + KEEP_VIEWER + controls(snapshot, False) + (KEEP,)
try:
reply = gemini_reply(
TASK_PROMPTS["show"].format(topic=topic, max_slides=MAX_VIEWER_SLIDES),
history,
retrieve_slides(slide_query("show", topic, snapshot)),
)
except TutorError as err:
note = str(err)
except Exception:
log.exception("Unexpected error while choosing a slide")
note = "Something went wrong while looking for a slide. Please try again."
with session.lock:
st = session.state
if reply is None:
session.messages = session.messages + [{"role": "assistant", "content": note}]
_save(session, [{"role": "assistant", "content": note, "mode": "show", "topic": topic, "source": "error"}])
elif reply.slides:
answer = _verify_citations(reply.answer)
picked = _relevant_slides(reply.slides, answer) or reply.slides
st["slides"] = [[s.lecture, s.slide] for s in picked]
st["slide_idx"] = 0
content = _with_source(answer, _slides_note(answer))
session.messages = session.messages + [{"role": "assistant", "content": content}]
_save(session, [{"role": "assistant", "content": content, "mode": "show", "topic": topic, "source": "slides"}])
else:
st["no_slide_topic"] = st["topic"]
_bump(session)
finally:
with session.lock:
session.busy = False
if reply is None:
with session.lock:
messages, snapshot, version = list(session.messages), dict(session.state), session.version
yield (messages,) + viewer_outputs(snapshot) + controls(snapshot, True) + (version,)
return
with session.lock:
st = session.state
note = None
if not st["slides"]:
note = "No slide in the lecture files illustrates this topic."
else:
current = st["slides"][st["slide_idx"]]
if st.get("view") == current and st.get("mode") == "related":
if len(st["slides"]) == 1:
note = "The viewer already shows the only slide chosen for this answer."
st["slide_idx"] = (st["slide_idx"] + 1) % len(st["slides"])
set_view(st, *st["slides"][st["slide_idx"]], mode="related")
if note:
session.messages = session.messages + [{"role": "assistant", "content": note}]
_save(session, [{"role": "assistant", "content": note, "mode": "show", "topic": st["topic"], "source": "none"}])
messages, snapshot, version = list(session.messages), dict(st), _bump(session)
yield (messages,) + viewer_outputs(snapshot) + controls(snapshot, True) + (version,)
def on_step(step: int, session: Session) -> tuple:
with session.lock:
st = session.state
view = st.get("view")
if view:
slides = st.get("slides") or []
if st.get("mode") == "related" and view in slides:
st["slide_idx"] = (slides.index(view) + step) % len(slides)
set_view(st, *slides[st["slide_idx"]], mode="related")
else:
lecture, number = view
count = len(LECTURES[lecture].slides)
set_view(st, lecture, (number - 1 + step) % count + 1, mode="lecture")
snapshot = dict(st)
_track(session, "previous" if step < 0 else "next", {"view": snapshot.get("view")})
return viewer_outputs(snapshot)
def on_prev(request: gr.Request):
return on_step(-1, get_session(request))
def on_next(request: gr.Request):
return on_step(1, get_session(request))
def on_browse_lecture(lecture, request: gr.Request):
session = get_session(request)
with session.lock:
if lecture in LECTURES:
set_view(session.state, int(lecture), 1, mode="lecture")
snapshot = dict(session.state)
_track(session, "browse_lecture", {"lecture": lecture if lecture in LECTURES else None})
return viewer_outputs(snapshot)
def on_browse_slide(lecture, number, request: gr.Request):
session = get_session(request)
with session.lock:
st = session.state
if lecture not in LECTURES:
if not st.get("view"):
return viewer_outputs(dict(st), caption="Choose a lecture first, then type a slide number.")
lecture = st["view"][0]
count = len(LECTURES[lecture].slides)
set_view(st, int(lecture), max(1, min(int(number or 1), count)), mode="lecture")
snapshot = dict(st)
_track(session, "browse_slide", {"lecture": int(lecture), "slide": snapshot["view"][1]})
return viewer_outputs(snapshot)
def on_example(request: gr.Request):
_track(get_session(request), "example")
def on_clear(request: gr.Request):
session = get_session(request)
_track(session, "clear")
with session.lock:
if session.busy:
return NOOP_TURN
session.messages = []
session.state = new_state()
snapshot, version = dict(session.state), _bump(session)
return ([], "") + viewer_outputs(snapshot) + controls(snapshot, True) + (version,)
# --------------------------------------------------------------------------- #
# Sign-in, accounts, and time on site
# --------------------------------------------------------------------------- #
def _login_problem(message: str) -> tuple:
return (gr.update(visible=True), gr.update(visible=False), message) + (KEEP,) * 17
def _store_problem(error: str) -> str:
log.error("Supabase: %s", error)
return "Sign-in is unavailable right now: the account database could not be reached. Please try again in a minute."
def _validate_login(user_id: str, email: str) -> str:
if not is_valid_user_id(user_id):
return "Enter a user ID of 2 to 64 characters: letters, digits, dots, underscores, or hyphens."
if not is_valid_email(email):
return "Enter a valid email address."
if not STORE.enabled:
return "Sign-in is not configured. " + STORE.config_hint()
return ""
def _restore(rows: list[dict]) -> tuple[list[dict], dict]:
"""Chat messages and tutor state rebuilt from a user's stored messages."""
messages = [
{"role": row["role"], "content": row["content"]}
for row in rows
if row.get("role") in ("user", "assistant") and isinstance(row.get("content"), str)
]
state = new_state()
# Mirror what a live turn records: failed turns and Show Image notes never enter the Gemini history or set the topic.
answered = []
pending_user = None
for row in rows:
if row.get("role") not in ("user", "assistant") or not isinstance(row.get("content"), str) or row.get("mode") == "show":
continue
if row["role"] == "user":
pending_user = row
elif row.get("source") != "error":
answered += [pending_user, row] if pending_user is not None else [row]
pending_user = None
else:
pending_user = None
history = []
for row in answered:
body, _ = _split_source(row["content"])
history.append({"role": "user" if row["role"] == "user" else "model", "text": body})
state["history"] = history[-2 * MAX_HISTORY_TURNS:]
for row in reversed(answered):
if row.get("role") == "assistant" and row.get("topic"):
state["topic"] = row["topic"]
break
for row in reversed(answered):
if row.get("role") == "assistant" and row.get("mode") != "quiz" and row.get("source") in ("slides", "web", "slides+web", "none"):
state["last_answer"], note = _split_source(row["content"])
state["last_note"] = note.replace(OLD_TUTOR_NOTE, "").strip() # notes saved by the previous version
state["last_source"] = row["source"]
break
return messages, state
def _begin(session: Session, user_id: str, email: str, rows: list[dict]) -> None:
with session.lock:
session.signed_out = False
previous = session.session_id if not session.ended else None
session.user_id, session.email, session.session_id = user_id, email, uuid.uuid4().hex
session.ended, session.last_beat = False, time.monotonic()
session.messages, session.state = _restore(rows)
_bump(session)
session_id = session.session_id
if previous:
STORE.enqueue("end_session", previous)
STORE.enqueue("start_session", session_id, user_id)
def _badge(session: Session, note: str = "") -> str:
badge = f"Signed in as **{session.user_id}** ({session.email})."
return f"{badge} {note}" if note else badge
def _enter_app(session: Session, restored: int, warning: str = "") -> tuple:
with session.lock:
messages, snapshot, version = list(session.messages), dict(session.state), session.version
badge = _badge(session, f"{restored} earlier messages restored." if restored else "")
if warning:
badge += " " + warning
return (gr.update(visible=False), gr.update(visible=True), "", badge, gr.Timer(active=True), messages, "") \
+ viewer_outputs(snapshot) + controls(snapshot, True) + (KEEP, KEEP, version)
def on_sign_in(user_id, email, request: gr.Request):
problem = _validate_login(user_id, email)
if problem:
return _login_problem(problem)
user_id, email = normalize_user_id(user_id), normalize_email(email)
session = get_session(request)
with session.lock:
if session.busy:
return _login_problem("Please wait for the current answer to finish.")
session.epoch += 1
row, error = STORE.get_user(user_id)
if error:
return _login_problem(_store_problem(error))
if row is None:
return _login_problem("No account has this user ID. Check the spelling, or choose Create account.")
if row.get("email") != email:
return _login_problem("The email does not match the one saved for this user ID.")
_settle_other_tabs(user_id, session)
rows, error = STORE.load_messages(user_id, RESTORE_MESSAGES)
warning = ""
if error:
log.error("Supabase: %s", error)
warning = "Earlier messages could not be loaded."
_begin(session, user_id, email, rows)
STORE.enqueue("record_login", user_id)
_track(session, "sign_in")
return _enter_app(session, len(rows), warning)
def on_create_account(user_id, email, request: gr.Request):
problem = _validate_login(user_id, email)
if problem:
return _login_problem(problem)
user_id, email = normalize_user_id(user_id), normalize_email(email)
session = get_session(request)
with session.lock:
if session.busy:
return _login_problem("Please wait for the current answer to finish.")
session.epoch += 1
_, error = STORE.create_user(user_id, email)
if error == EXISTS:
return _login_problem("This user ID is already taken. Sign in with it, or choose another ID.")
if error:
return _login_problem(_store_problem(error))
_begin(session, user_id, email, [])
_track(session, "create_account")
return _enter_app(session, 0)
def on_sign_out(request: gr.Request):
session = get_session(request)
with session.lock:
busy = session.busy
if busy:
gr.Warning("Please wait until the current answer is finished, then sign out.")
return NOOP_LOGIN
with session.lock:
user_id, session_id, ended = session.user_id, session.session_id, session.ended
session.user_id = session.email = session.session_id = None
session.ended = False
session.signed_out = True
session.epoch += 1
session.messages = []
session.state = new_state()
snapshot = dict(session.state)
_bump(session)
if user_id and session_id:
STORE.enqueue("log_event", user_id, session_id, "sign_out", {})
if not ended:
STORE.enqueue("end_session", session_id)
# Clear the sign-in fields for the next student; version 0 tells the page's sync check that nobody is signed in.
return (gr.update(visible=True), gr.update(visible=False), "", "", gr.Timer(active=False), [], "") \
+ viewer_outputs(snapshot) + controls(snapshot, True) + ("", "", 0)
def _other_tabs(user_id: str, current: Session) -> list[Session]:
with _sessions_lock:
return [s for s in SESSIONS.values() if s is not current and s.user_id == user_id]
def _answer_in_progress(user_id: str, current: Session) -> bool:
"""Another tab of the same student is still writing an answer (a reloaded page signs in while the answer its
old tab asked for may still be in progress)."""
return any(s.busy for s in _other_tabs(user_id, current))
def _settle_other_tabs(user_id: str, current: Session) -> None:
"""Make sure answers that other tabs of this student saved recently have reached Supabase before the chat is
loaded. Waits for the shared write queue only when there is such an answer, so ordinary sign-ins never wait."""
recent = time.monotonic() - 2 * HEARTBEAT_SECONDS
if any(s.saved_at > recent for s in _other_tabs(user_id, current)):
STORE.flush(10.0)
def _resume(session: Session, user_id, email, version) -> bool:
"""Sign a student back in without the sign-in form: after the Space restarted while the page stayed open, or
when the page reloaded itself after a broken connection (on_page_load). The page sends the user ID and email it
signed in with and the chat version it shows; version 0 means the page shows the sign-in form, which is never
signed in this way, and neither is a tab whose student pressed Sign out. The same check as Sign in applies:
the user ID must exist and the email must match. Returns True when the student is signed in afterwards."""
try:
shown = int(version or 0)
except (TypeError, ValueError):
shown = 0
if shown <= 0 or not STORE.enabled or not is_valid_user_id(user_id) or not is_valid_email(email):
return False
user_id, email = normalize_user_id(user_id), normalize_email(email)
with session.resume_lock:
with session.lock:
if session.user_id is not None:
return session.user_id == user_id
if session.signed_out:
return False
epoch = session.epoch
row, error = STORE.get_user(user_id)
if error or row is None or row.get("email") != email:
if error:
log.error("Supabase: %s", error)
return False
_settle_other_tabs(user_id, session)
rows, error = STORE.load_messages(user_id, RESTORE_MESSAGES)
if error: # try again on the next check rather than show an empty chat
log.error("Supabase: %s", error)
return False
with session.lock: # Sign in, Create account, or Sign out happened meanwhile: they win
if session.user_id is not None or session.signed_out or session.epoch != epoch:
return session.user_id == user_id
_begin(session, user_id, email, rows)
log.info("Signed a student back in automatically")
_track(session, "resume")
return True
def _beat(session: Session) -> None:
"""Write the visit's time on the site to Supabase at most every HEARTBEAT_SECONDS."""
now = time.monotonic()
with session.lock:
if not session.session_id or session.ended or now - session.last_beat < HEARTBEAT_SECONDS - 1:
return
session.last_beat, session_id = now, session.session_id
STORE.enqueue("heartbeat", session_id)
def on_sync(version, user_id, email, request: gr.Request):
"""Bring an open, signed-in page up to date with the server.
Runs every SYNC_SECONDS, when the tab becomes visible again, when the browser comes back online, and when
Gradio reports that the connection was re-established. If the page's connection dropped while an answer was
being written, the page never received that answer and its buttons stayed disabled: the chat version it shows
is then older than the server's, and this sends the current chat, viewer, and buttons. After a server restart
the student is signed back in first (see _resume). Otherwise nothing on the page changes.
"""
try:
shown = int(version or 0)
except (TypeError, ValueError):
shown = 0
if shown <= 0: # the page shows the sign-in form
return NOOP_SYNC
session = get_session(request)
badge = KEEP
if session.user_id is None:
if not _resume(session, user_id, email, version):
return NOOP_SYNC
badge = _badge(session, "The site restarted; your chat was restored.")
_reopen(session)
_beat(session)
_collect_awaited_answer(session)
with session.lock:
if session.busy or session.version == shown:
return NOOP_SYNC
messages, snapshot, current = list(session.messages), dict(session.state), session.version
return (messages, KEEP, current, badge) + viewer_outputs(snapshot) + controls(snapshot, True)
def _collect_awaited_answer(session: Session) -> None:
"""A reloaded page signed in while its old tab was still writing an answer: once that answer is saved, reload
the chat from Supabase so that this page shows it too (on_sync then sends it)."""
with session.lock:
user_id = session.user_id if session.awaiting and not session.busy else None
if not user_id or _answer_in_progress(user_id, session):
return
_settle_other_tabs(user_id, session)
rows, error = STORE.load_messages(user_id, RESTORE_MESSAGES)
if error:
log.error("Supabase: %s", error)
return
with session.lock:
if session.awaiting and session.user_id == user_id and not session.busy:
session.messages, session.state = _restore(rows)
session.awaiting = False
_bump(session)
def on_page_load(user_id, email, draft, request: gr.Request):
"""Sign the student back in when the page reloads itself after a broken connection.
Gradio's client receives no further results once its connection broke during a request, which is what froze
the page; SYNC_HEAD then reloads the page. The tab remembers the user ID and email it signed in with
(sessionStorage: this tab only, removed on Sign out), and LOAD_JS hands them over only right after that
automatic reload, so a manual reload, a new tab, or a closed tab reopened by the next person on a shared
computer starts at the sign-in form. An answer that the old page was still waiting for is finished first, then
the chat is shown from Supabase together with the question the student was typing.
"""
if not (user_id or email) or _validate_login(user_id, email):
yield NOOP_LOGIN
return
waiting = (KEEP, KEEP, "Reconnecting and restoring your chat...") + (KEEP,) * 17
yield waiting
session = get_session(request)
user = normalize_user_id(user_id)
deadline = time.monotonic() + ANSWER_WAIT_SECONDS
while _answer_in_progress(user, session) and time.monotonic() < deadline:
time.sleep(1.0) # Gradio frees the worker thread between steps of a generator
yield waiting
# Questions that an old tab is still answering: they will be shown when saved (see _collect_awaited_answer).
in_progress = {
s.messages[-1]["content"].strip()
for s in _other_tabs(user, session) if s.busy and s.messages and s.messages[-1]["role"] == "user"
}
resumed = _resume(session, user_id, email, 1)
with session.lock:
someone_else = session.user_id is not None and session.user_id != user
if someone_else: # another student signed in on this tab meanwhile: leave it to them
yield NOOP_LOGIN
return
if not resumed:
yield (gr.update(visible=True), gr.update(visible=False),
"Your chat could not be restored automatically. Please sign in again.") + (KEEP,) * 17
return
with session.lock:
session.awaiting = bool(in_progress)
out = list(_enter_app(session, 0))
with session.lock:
out[3] = _badge(session, f"Reconnected; {len(session.messages)} messages restored." if session.messages else "Reconnected.")
asked = {m["content"].strip() for m in session.messages if m["role"] == "user"} | in_progress
# The question being typed when the page reloaded, or the question whose answer was still being written;
# left out when the restored chat already contains it or the old tab is still answering it.
out[6] = "" if (draft or "").strip() in asked else (draft or "")
out[17], out[18] = user, normalize_email(email)
yield tuple(out)
# --------------------------------------------------------------------------- #
# Interface
# --------------------------------------------------------------------------- #
def _logo_html() -> str:
if LOGO_PATH is None:
return ""
with Image.open(LOGO_PATH) as source:
logo = source.convert("RGBA")
bbox = logo.getchannel("A").getbbox()
if bbox:
logo = logo.crop(bbox) # drop the transparent margins around the mark
logo.thumbnail((480, 120), Image.LANCZOS) # shown at 28px; keep enough pixels for dense screens
buffer = io.BytesIO()
logo.save(buffer, format="PNG", optimize=True)
data = base64.b64encode(buffer.getvalue()).decode("ascii")
return f'<img src="data:image/png;base64,{data}" alt="Creighton Bluejays logo">'
HEADER_HTML = """
<div id="header">
<div>
<h1>Creighton Anatomy Learning Platform</h1>
<p>Answers, quizzes, and clinical correlations grounded in the course lecture slides.</p>
</div>
</div>
"""
FOOTER_HTML = f'<div id="footer">{_logo_html()}</div>'
PAGE_BACKGROUND = "linear-gradient(180deg, #F7FAFD 0%, #E9F1F9 100%)"
CARD_BORDER = "#C9DDF0"
CARD_SHADOW = "0 4px 14px rgba(0, 35, 93, 0.05)"
SERIF = "'Playfair Display', Georgia, 'Times New Roman', serif"
CSS = f"""
body {{ background: {PAGE_BACKGROUND} !important; }}
.gradio-container {{ max-width: 1400px !important; margin: 0 auto !important; background: transparent !important; color: {CU_NAVY}; }}
#header {{
background: linear-gradient(135deg, #FFFFFF 0%, {CU_TINT} 100%); color: {CU_NAVY};
border: 1px solid {CARD_BORDER}; border-top: 4px solid {CU_BLUE}; border-radius: 12px;
padding: 22px 28px; margin-bottom: 8px; box-shadow: 0 6px 18px rgba(0, 35, 93, 0.06);
}}
#header h1 {{
margin: 0; font-family: {SERIF}; font-size: 1.9rem; font-weight: 600; line-height: 1.2;
color: {CU_NAVY}; letter-spacing: 0.2px;
}}
#header p {{ margin: 6px 0 0; color: {CU_BLUE}; font-size: 1rem; font-weight: 500; }}
#login-panel {{
max-width: 520px; margin: 24px auto; padding: 26px 28px; background: #FFFFFF;
border: 1px solid {CARD_BORDER}; border-radius: 12px; box-shadow: {CARD_SHADOW};
}}
#login-panel h3 {{ margin: 0 0 4px; color: {CU_BLUE}; }}
#login-msg {{ color: #B3261E; font-weight: 600; min-height: 1.4em; }}
#account-bar {{ align-items: center; margin-bottom: 4px; }}
#account-bar .prose {{ color: {CU_NAVY}; font-size: 0.95rem; }}
#chat {{ border: 1px solid {CARD_BORDER}; border-radius: 12px; box-shadow: {CARD_SHADOW}; }}
#chat .message.bot {{ background: #FFFFFF; border: 1px solid #D7E6F4; color: {CU_NAVY}; }}
#chat .message.user {{ background: {CU_TINT}; border: 1px solid {CARD_BORDER}; color: {CU_NAVY}; }}
#chat .placeholder-content {{ color: {CU_BLUE}; }}
#chat .prose {{ color: {CU_NAVY}; font-size: 1.02rem; line-height: 1.6; }}
#chat .prose h1, #chat .prose h2, #chat .prose h3, #chat .prose h4, #chat .prose summary {{ color: {CU_BLUE}; }}
#chat .prose strong {{ color: {CU_NAVY}; }}
#chat .prose a {{ color: {CU_BLUE}; }}
#quick-actions button {{
border: 1px solid {CU_BLUE}; color: {CU_BLUE}; background: #FFFFFF; font-weight: 600; font-size: 0.95rem;
padding-left: 10px; padding-right: 10px; min-width: 0; white-space: nowrap;
}}
#quick-actions button:hover:not([disabled]) {{ background: {CU_TINT}; }}
#quick-actions button[disabled] {{ border-color: {CARD_BORDER}; color: #8FB0D3; background: #F7FAFD; }}
#viewer-title {{
color: {CU_BLUE}; font-weight: 700; font-size: 0.8rem; text-transform: uppercase; letter-spacing: 0.09em; margin: 6px 0;
}}
#slide-image {{ border: 1px solid {CARD_BORDER}; border-radius: 12px; background: #FFFFFF; box-shadow: {CARD_SHADOW}; }}
#slide-caption {{ color: {CU_NAVY}; font-size: 0.95rem; min-height: 3.2em; }}
#footer {{ display: flex; justify-content: center; padding: 30px 0 12px; }}
#footer img {{ height: 28px; width: auto; opacity: 0.55; }}
footer {{ display: none !important; }}
#sync-btn {{ display: none !important; }}
"""
# Runs the page's sync check (on_sync) as soon as the student is back: when the tab becomes visible again, when the
# browser comes back online, when the page is restored from the back/forward cache, and when Gradio shows its
# "Connection re-established" message. The periodic check every SYNC_SECONDS covers everything else.
SYNC_HEAD = """
<script>
(function () {
var last = 0;
function sync() {
var button = document.getElementById("sync-btn");
if (!button || Date.now() - last < 3000) return;
last = Date.now();
button.click();
}
function soon(ms) { setTimeout(sync, ms); }
// Gradio's client receives no further results once its connection broke while a request was running (the
// frozen page). When Gradio reports that the server is reachable again, reload the page: it signs the student
// back in by itself (on_page_load) and keeps the question being typed.
function reload() {
try {
var since = Date.now() - Number(sessionStorage.getItem("anatomy-reloaded") || 0);
if (since < 10000) { setTimeout(reload, 10000 - since + 100); return; } // at most one reload per 10 s
sessionStorage.setItem("anatomy-reloaded", String(Date.now()));
sessionStorage.setItem("anatomy-auto-sign-in", String(Date.now())); // read once by LOAD_JS
var box = document.querySelector("#question textarea");
var draft = box ? box.value : "";
// A question whose answer was still being written ("Thinking") is put back in the question box after the
// reload, unless the restored chat already contains it (see on_page_load).
var chat = document.getElementById("chat");
if (!draft && chat && /Thinking/.test(chat.innerText)) {
var asked = chat.querySelectorAll(".message.user, [data-testid='user']");
var last = asked.length ? asked[asked.length - 1].innerText.trim() : "";
if (last && !/^(Quiz me on: |Clinical correlation for: |Simplify the previous explanation)/.test(last)) draft = last;
}
var who = JSON.parse(sessionStorage.getItem("anatomy-sign-in") || "null");
if (draft && who) sessionStorage.setItem("anatomy-draft", JSON.stringify({user_id: who.user_id, text: draft}));
} catch (e) {}
location.reload();
}
document.addEventListener("visibilitychange", function () {
if (document.visibilityState === "visible") soon(800);
});
window.addEventListener("online", function () { soon(2000); });
window.addEventListener("pageshow", function (event) { if (event.persisted) soon(800); });
new MutationObserver(function (records) {
for (var i = 0; i < records.length; i++) {
var added = records[i].addedNodes;
for (var j = 0; j < added.length; j++) {
var node = added[j];
if (node.nodeType === 1 && (node.matches(".toast-body") || node.querySelector(".toast-body")) &&
/Connection re-established|server has changed|Session not found/.test(node.textContent)) {
setTimeout(reload, 500);
return;
}
}
}
}).observe(document.documentElement, {childList: true, subtree: true});
})();
</script>
"""
# The browser tab remembers who signed in (sessionStorage: this tab only), so that a reload signs in by itself.
REMEMBER_JS = """(user, email, version) => {
try {
if (version > 0) {
sessionStorage.setItem("anatomy-sign-in", JSON.stringify({user_id: user, email: email}));
// A question kept from before an automatic reload whose sign-in failed goes back into the question box,
// for the same student only.
var draft = JSON.parse(sessionStorage.getItem("anatomy-draft") || "null");
var box = document.querySelector("#question textarea");
var same = draft && String(draft.user_id).trim().toLowerCase() === String(user).trim().toLowerCase();
if (same && draft.text && box && !box.value) { box.value = draft.text; box.dispatchEvent(new Event("input", {bubbles: true})); }
sessionStorage.removeItem("anatomy-draft");
} else sessionStorage.removeItem("anatomy-sign-in");
} catch (e) {}
return [];
}"""
# After the reloaded page signed in by itself, the kept draft has been put back (on_page_load) and is dropped.
LOADED_JS = """(version) => { try { if (version > 0) sessionStorage.removeItem("anatomy-draft"); } catch (e) {} return []; }"""
FORGET_JS = """(version) => {
try {
if (!(version > 0)) { sessionStorage.removeItem("anatomy-sign-in"); sessionStorage.removeItem("anatomy-draft"); }
} catch (e) {}
return [];
}"""
# Only a page that reloaded itself after a broken connection (marker set by SYNC_HEAD's reload, at most a minute
# old) signs in by itself; a manual reload or a reopened tab shows the sign-in form.
LOAD_JS = """() => {
try {
var marker = Number(sessionStorage.getItem("anatomy-auto-sign-in") || 0);
sessionStorage.removeItem("anatomy-auto-sign-in");
var draft = JSON.parse(sessionStorage.getItem("anatomy-draft") || "null");
var saved = JSON.parse(sessionStorage.getItem("anatomy-sign-in") || "null");
var text = draft && saved && draft.user_id === saved.user_id ? draft.text : "";
if (saved && saved.user_id && Date.now() - marker < 60000) return [saved.user_id, saved.email, text];
} catch (e) {}
return ["", "", ""];
}"""
THEME = gr.themes.Base(
primary_hue=gr.themes.Color(
c50=CU_TINT, c100="#D4E6F6", c200="#A9CCEC", c300=CU_LIGHT_BLUE, c400="#3A87C9",
c500=CU_BLUE, c600="#004F92", c700="#00417A", c800="#003262", c900=CU_NAVY, c950="#001A45",
),
secondary_hue=gr.themes.colors.sky,
neutral_hue=gr.themes.colors.slate,
font=[
gr.themes.GoogleFont("Source Sans 3"),
gr.themes.GoogleFont("Playfair Display"), # loaded for the page title
"ui-sans-serif", "system-ui", "sans-serif",
],
radius_size=gr.themes.sizes.radius_lg,
).set(
body_background_fill=PAGE_BACKGROUND,
body_background_fill_dark=PAGE_BACKGROUND,
body_text_color=CU_NAVY,
body_text_color_dark=CU_NAVY,
body_text_color_subdued="#4E7CB0",
body_text_color_subdued_dark="#4E7CB0",
background_fill_primary="#FFFFFF",
background_fill_primary_dark="#FFFFFF",
background_fill_secondary="#F5F8FB",
background_fill_secondary_dark="#F5F8FB",
block_background_fill="#FFFFFF",
block_background_fill_dark="#FFFFFF",
block_border_color=CARD_BORDER,
block_border_color_dark=CARD_BORDER,
block_shadow=CARD_SHADOW,
block_shadow_dark=CARD_SHADOW,
block_title_text_color=CU_BLUE,
block_title_text_color_dark=CU_BLUE,
block_title_text_weight="600",
block_label_text_color=CU_BLUE,
block_label_text_color_dark=CU_BLUE,
block_label_text_weight="600",
block_label_background_fill="#FFFFFF",
block_label_background_fill_dark="#FFFFFF",
border_color_primary=CARD_BORDER,
border_color_primary_dark=CARD_BORDER,
border_color_accent=CU_LIGHT_BLUE,
border_color_accent_dark=CU_LIGHT_BLUE,
border_color_accent_subdued="#B9D6EE",
border_color_accent_subdued_dark="#B9D6EE",
color_accent=CU_BLUE,
color_accent_soft=CU_TINT,
color_accent_soft_dark=CU_TINT,
link_text_color=CU_BLUE,
link_text_color_dark=CU_BLUE,
input_background_fill="#FFFFFF",
input_background_fill_dark="#FFFFFF",
input_background_fill_focus="#FFFFFF",
input_background_fill_focus_dark="#FFFFFF",
input_border_color=CARD_BORDER,
input_border_color_dark=CARD_BORDER,
input_border_color_focus=CU_BLUE,
input_border_color_focus_dark=CU_BLUE,
input_placeholder_color="#7FA3CB",
input_placeholder_color_dark="#7FA3CB",
button_primary_background_fill=CU_BLUE,
button_primary_background_fill_dark=CU_BLUE,
button_primary_background_fill_hover=CU_NAVY,
button_primary_background_fill_hover_dark=CU_NAVY,
button_primary_text_color="#FFFFFF",
button_primary_text_color_dark="#FFFFFF",
button_primary_border_color=CU_BLUE,
button_primary_border_color_dark=CU_BLUE,
button_secondary_background_fill="#FFFFFF",
button_secondary_background_fill_dark="#FFFFFF",
button_secondary_background_fill_hover=CU_TINT,
button_secondary_background_fill_hover_dark=CU_TINT,
button_secondary_text_color=CU_BLUE,
button_secondary_text_color_dark=CU_BLUE,
button_secondary_border_color=CU_BLUE,
button_secondary_border_color_dark=CU_BLUE,
table_border_color=CU_LIGHT_GRAY,
table_border_color_dark=CU_LIGHT_GRAY,
table_even_background_fill="#F5F8FB",
table_even_background_fill_dark="#F5F8FB",
table_odd_background_fill="#FFFFFF",
table_odd_background_fill_dark="#FFFFFF",
code_background_fill="#F5F8FB",
code_background_fill_dark="#F5F8FB",
)
LECTURE_CHOICES = [(lec.label, lec.number) for lec in LECTURES.values()]
EXAMPLES = [
"Which nerve roots form the brachial plexus?",
"What forms the superior thoracic aperture?",
"Name the parts of the sternum.",
]
with gr.Blocks(title="Creighton Anatomy Learning Platform", theme=THEME, css=CSS, head=SYNC_HEAD) as demo:
gr.HTML(HEADER_HTML)
with gr.Column(elem_id="login-panel") as login_panel:
gr.Markdown(
"### Sign in\nEnter your user ID and email. New here? Fill in both and choose **Create account**.\n\n"
"Your questions, answers, time on the site, and button use are saved with your user ID so that you can "
"return to your chat and the course team can improve the platform."
)
login_user = gr.Textbox(label="User ID", placeholder="e.g. jdoe", max_lines=1)
login_email = gr.Textbox(label="Email", placeholder="you@creighton.edu", max_lines=1)
with gr.Row():
signin_btn = gr.Button("Sign in", variant="primary")
signup_btn = gr.Button("Create account", variant="secondary")
login_msg = gr.Markdown("", elem_id="login-msg")
with gr.Column(visible=False) as app_panel:
with gr.Row(elem_id="account-bar"):
with gr.Column(scale=6):
user_badge = gr.Markdown("")
with gr.Column(scale=1, min_width=110):
signout_btn = gr.Button("Sign out", size="sm", variant="secondary")
with gr.Row(equal_height=False):
with gr.Column(scale=3, elem_id="chat-col"):
chatbot = gr.Chatbot(
type="messages",
height=540,
show_label=False,
elem_id="chat",
placeholder="Ask about any structure, region, or concept covered in the lectures.",
)
with gr.Row():
question = gr.Textbox(
elem_id="question",
placeholder="Ask a question about the lecture material",
show_label=False,
container=False,
lines=1,
max_lines=4,
scale=6,
autofocus=True,
)
ask_btn = gr.Button("Ask", variant="primary", scale=1, min_width=90)
clear_btn = gr.Button("Clear", variant="secondary", scale=1, min_width=90)
with gr.Row(elem_id="quick-actions"):
show_btn = gr.Button("Show Image", interactive=False)
quiz_btn = gr.Button("Quiz Me", interactive=False)
clinical_btn = gr.Button("Clinical Correlation", interactive=False)
simplify_btn = gr.Button("Simplify", interactive=False)
examples = gr.Examples(examples=EXAMPLES, inputs=question, label="Try asking")
with gr.Column(scale=2, elem_id="viewer-col"):
gr.HTML('<div id="viewer-title">Slide Viewer</div>')
slide_img = gr.Image(
type="filepath",
show_label=False,
height=400,
interactive=False,
show_download_button=False,
show_fullscreen_button=True,
elem_id="slide-image",
)
slide_caption = gr.Markdown(_caption(new_state()), elem_id="slide-caption")
with gr.Row():
prev_btn = gr.Button("Previous", size="sm")
next_btn = gr.Button("Next", size="sm")
lecture_dd = gr.Dropdown(choices=LECTURE_CHOICES, label="Browse a lecture", value=None)
slide_num = gr.Number(value=1, precision=0, label="Go to slide", info="Type a slide number and press Enter")
gr.HTML(FOOTER_HTML)
heartbeat = gr.Timer(SYNC_SECONDS, active=False)
# The chat version the page shows (0: the sign-in form). Hidden; sent with every request so the server can tell
# whether the page missed an update.
page_version = gr.Number(value=0, precision=0, visible=False)
sync_btn = gr.Button("Sync", elem_id="sync-btn") # hidden by CSS; clicked by SYNC_HEAD
viewer = [slide_img, slide_caption, lecture_dd, slide_num]
control_btns = [show_btn, quiz_btn, clinical_btn, simplify_btn, ask_btn, clear_btn]
creds = [login_user, login_email, page_version]
turn_outputs = [chatbot, question] + viewer + control_btns + [page_version]
show_outputs = [chatbot] + viewer + control_btns + [page_version]
login_outputs = [login_panel, app_panel, login_msg, user_badge, heartbeat, chatbot, question] + viewer + control_btns \
+ [login_user, login_email, page_version]
sync_outputs = [chatbot, question, page_version, user_badge] + viewer + control_btns
remembered = [login_user, login_email, page_version]
signin_btn.click(on_sign_in, [login_user, login_email], login_outputs).then(None, remembered, None, js=REMEMBER_JS)
login_user.submit(on_sign_in, [login_user, login_email], login_outputs).then(None, remembered, None, js=REMEMBER_JS)
login_email.submit(on_sign_in, [login_user, login_email], login_outputs).then(None, remembered, None, js=REMEMBER_JS)
signup_btn.click(on_create_account, [login_user, login_email], login_outputs).then(None, remembered, None, js=REMEMBER_JS)
signout_btn.click(on_sign_out, None, login_outputs).then(None, [page_version], None, js=FORGET_JS)
demo.load(on_page_load, [login_user, login_email, question], login_outputs, js=LOAD_JS, concurrency_limit=None) \
.then(None, [page_version], None, js=LOADED_JS)
sync_inputs = [page_version, login_user, login_email]
heartbeat.tick(on_sync, sync_inputs, sync_outputs, show_progress="hidden", concurrency_limit=None)
sync_btn.click(on_sync, sync_inputs, sync_outputs, show_progress="hidden", concurrency_limit=None)
ask_btn.click(on_ask, [question] + creds, turn_outputs)
question.submit(on_ask, [question] + creds, turn_outputs)
quiz_btn.click(on_quiz, creds, turn_outputs)
clinical_btn.click(on_clinical, creds, turn_outputs)
simplify_btn.click(on_simplify, creds, turn_outputs)
show_btn.click(on_show_image, creds, show_outputs)
clear_btn.click(on_clear, None, turn_outputs)
prev_btn.click(on_prev, None, viewer)
next_btn.click(on_next, None, viewer)
lecture_dd.input(on_browse_lecture, [lecture_dd], viewer)
slide_num.submit(on_browse_slide, [lecture_dd, slide_num], viewer)
if getattr(examples, "load_input_event", None) is not None:
examples.load_input_event.then(on_example, None, None)
demo.unload(forget_session)
web_sources.warm_up()
atexit.register(STORE.flush)
if __name__ == "__main__":
demo.queue(default_concurrency_limit=4).launch(ssr_mode=False)