VisualStep-app / app.py
lsadouk1111's picture
Update app.py
b21d86c verified
Raw History Blame Contribute Delete
10.8 kB
"""
VisualStep β€” ADHD Classroom Instruction Decomposer
Gradio demo app for Hugging Face Spaces
"""
import json
import gradio as gr
import torch
import spaces
# ── Model loading ─────────────────────────────────────────────────────────────
MODEL_ID = "lsadouk1111/VisualStep"
model = None
tokenizer = None
# ── System prompt ─────────────────────────────────────────────────────────────
SYSTEM_PROMPT = """You are an expert in ADHD classroom accommodations for primary school children aged 5-10.
Transform the teacher's spoken classroom instruction into an ADHD-adapted visual step card.
Rules:
- Maximum 5 steps
- Each step: one concrete physical action only
- Each action: maximum 6 words
- Start each step with an imperative verb
- Include a pictogram keyword in the icon field
- Output JSON only, no other text
Output format:
{"steps":[{"id":1,"action":"Open your book","detail":"page 23","icon":"book","check":true}],"n_steps":1,"support_level":"medium"}
support_level: "light" (1-2 steps), "medium" (3-4 steps), "high" (5 steps)"""
# ── Icon to emoji ──────────────────────────────────────────────────────────────
EMOJI = {
# core actions
"book": "πŸ“–", "page": "πŸ“„", "pencil": "✏️", "write": "✏️", "writing": "✏️",
"read": "πŸ‘οΈ", "eye": "πŸ‘οΈ", "hand": "βœ‹", "finger": "πŸ‘†",
"talk": "πŸ—£οΈ", "listen": "πŸ‘‚", "think": "🧠", "look": "πŸ‘οΈ", "show": "πŸ‘οΈ",
"draw": "🎨", "paper": "πŸ“", "notebook": "πŸ““", "board": "πŸ“‹",
"number": "πŸ”’", "count": "πŸ”’", "answer": "βœ…", "sort": "πŸ—‚οΈ",
"match": "πŸ”—", "circle": "β­•", "plant": "🌱", "seed": "🌰",
"water": "πŸ’§", "animal": "🐾", "picture": "πŸ–ΌοΈ", "image": "πŸ–ΌοΈ", "photo": "πŸ–ΌοΈ",
"partner": "πŸ‘₯", "find": "πŸ”", "search": "πŸ”", "observe": "πŸ”­",
"measure": "πŸ“", "ruler": "πŸ“", "line": "πŸ“",
"record": "πŸ“Š", "bar": "πŸ“Š", "label": "🏷️",
"colour": "πŸ–ŒοΈ", "color": "πŸ–ŒοΈ", "paintbrush": "πŸ–ŒοΈ",
"cut": "βœ‚οΈ", "touch": "πŸ‘†", "quiet": "🀫", "ready": "βœ…", "check": "βœ…",
"open": "πŸ“‚", "put": "πŸ“Œ", "place": "πŸ“Œ", "sit": "πŸͺ‘", "chair": "πŸͺ‘",
"carpet": "πŸͺ‘", "desk": "πŸͺ‘",
"science": "πŸ”¬", "maths": "βž•", "english": "πŸ“š",
# people / social
"people": "πŸ‘₯", "two_people": "πŸ‘₯", "two-people": "πŸ‘₯", "friend": "πŸ‘₯",
"group": "πŸ‘₯", "class": "🏫", "classroom": "🏫", "pairs": "πŸ‘₯", "share": "πŸ‘₯",
# speech / thought
"pause": "⏸️", "ear": "πŸ‘‚", "speak": "πŸ—£οΈ", "discussion": "πŸ—£οΈ",
"buzz": "πŸ—£οΈ", "feedback": "πŸ’¬", "speech_bubble": "πŸ’¬", "speech-bubble": "πŸ’¬",
"thought": "πŸ’­", "thought-bubble": "πŸ’­", "thought_bubble": "πŸ’­",
# writing tools
"marker": "πŸ–ŠοΈ", "pen": "πŸ–ŠοΈ", "highlighter": "πŸ–ŠοΈ", "eraser": "✏️",
"edit": "✏️", "underline": "πŸ“", "sentence": "πŸ“", "word": "πŸ“",
"text": "πŸ“", "paragraph": "πŸ“", "title": "πŸ“", "story": "πŸ“–",
# maths
"math": "βž•", "equation": "βž•", "plus": "βž•", "add": "βž•", "addend": "βž•",
"divide": "βž—", "calculator": "πŸ”’", "blocks": "🧱", "build": "🧱",
"model": "🧱", "pattern": "πŸ”’", "seven": "7️⃣",
# organising / tasks
"select": "πŸ‘†", "choose": "πŸ‘†", "pick": "πŸ‘†",
"folder": "πŸ“", "task": "πŸ“‹", "tasks": "πŸ“‹", "list": "πŸ“‹", "work": "πŸ“‹",
"fact": "πŸ“‹", "options": "πŸ“‹", "details": "πŸ“‹", "rules": "πŸ“‹",
"examples": "πŸ“‹", "problem": "πŸ“‹", "compare": "πŸ”—", "checklist": "βœ…",
"solve": "βœ…", "questions": "❓", "question": "❓",
# ideas / feedback
"idea": "πŸ’‘", "lightbulb": "πŸ’‘", "decision": "πŸ’‘",
# time / waiting
"wait": "⏳", "clock": "⏰", "repeat": "πŸ”", "play": "▢️",
# objects / materials
"box": "πŸ“¦", "materials": "πŸ“¦", "coat": "πŸ§₯", "wire": "πŸ”Œ",
"cup": "πŸ₯›", "spoon": "πŸ₯„", "beaker": "πŸ§ͺ", "thermometer": "🌑️",
"worksheet": "πŸ“", "homework": "πŸ“", "exercise": "πŸ“",
# nature / misc
"tree": "🌳", "blueberry": "🫐", "nightshade": "🌿", "garden": "🌱",
"cloud": "☁️", "fire": "πŸ”₯", "lightning": "⚑", "explosion": "πŸ’₯",
"light": "πŸ’‘", "breathe": "🌬️", "matter": "πŸ”¬",
# feedback / emotion
"thumb": "πŸ‘", "happy": "😊", "heart": "❀️",
# navigation
"door": "πŸšͺ", "broken": "❌", "close": "❌",
# misc
"marble": "βšͺ", "period": "πŸ“", "photo": "πŸ–ΌοΈ",
}
def get_emoji(icon):
if not icon:
return "πŸ“Œ"
k = str(icon).lower()
for key, em in EMOJI.items():
if key in k:
return em
return "πŸ“Œ"
# ── Generation ────────────────────────────────────────────────────────────────
@spaces.GPU
def generate_step_card(instruction, year, subject):
global model, tokenizer
if not instruction.strip():
return "Please enter a classroom instruction."
# Load model on first call (inside GPU context)
if model is None:
from unsloth import FastModel
model, tokenizer = FastModel.from_pretrained(
model_name=MODEL_ID,
max_seq_length=512,
load_in_4bit=True,
dtype=None,
)
FastModel.for_inference(model)
model.eval()
messages = [
{"role": "system", "content": SYSTEM_PROMPT},
{"role": "user", "content": f"Year: {year} | Subject: {subject} | Instruction: {instruction}"},
]
prompt = tokenizer.apply_chat_template(
messages, tokenize=False, add_generation_prompt=True
)
inputs = tokenizer(prompt, return_tensors="pt").to(model.device)
with torch.no_grad():
outputs = model.generate(
**inputs,
max_new_tokens=256,
do_sample=False,
pad_token_id=tokenizer.eos_token_id,
)
generated = outputs[0][inputs["input_ids"].shape[1]:]
raw = tokenizer.decode(generated, skip_special_tokens=True).strip()
raw = raw.replace("```json", "").replace("```", "").strip()
try:
data = json.loads(raw)
steps = data.get("steps", [])
support = data.get("support_level", "medium")
n = data.get("n_steps", len(steps))
# Build visual card
lines = ["## πŸ“Œ WHAT TO DO NOW\n"]
for s in steps:
emoji = get_emoji(s.get("icon", ""))
action = s.get("action", "")
detail = s.get("detail", "")
line = f"☐ **{s.get('id', '')}.** {emoji} {action}"
if detail:
line += f" β†’ *{detail}*"
lines.append(line)
lines.append(f"\n---")
lines.append(f"*Support level: {support} Β· {n} step(s)*")
return "\n\n".join(lines)
except json.JSONDecodeError:
return f"⚠️ Model output could not be parsed as JSON.\n\nRaw output:\n```\n{raw}\n```"
# ── Gradio interface ───────────────────────────────────────────────────────────
EXAMPLES = [
["Ok everyone, open your books to page 23, read the paragraph quietly, then answer questions 1 to 3 in your notebook and put your hand up when you're done.", "2", "english"],
["Right, so, um, can we all take our science books and turn to page 10? And then, after that, I want you to look at the picture and think about what you see.", "1", "science"],
["Everyone stop what you're doing, pack away your things, and line up quietly at the door please.", "3", "routine"],
["I want you to sort these animals into mammals and not mammals using the sorting hoops, then write one sentence explaining how you decided.", "1", "science"],
["OK so, find your maths book, open to page 15, do the first three problems and check your answers with your partner.", "2", "maths"],
]
with gr.Blocks(
title="VisualStep β€” ADHD Instruction Decomposer",
theme=gr.themes.Soft(primary_hue="blue"),
css="""
.card-output { font-size: 1.1em; line-height: 1.8; }
h1 { color: #1F4E79; }
.subtitle { color: #666; font-size: 0.95em; margin-top: -10px; }
"""
) as demo:
gr.Markdown("""
# VisualStep
### ADHD-Adapted Visual Instruction Decomposition for Primary School Classrooms
*Fine-tuned Phi-3 Mini (3.8B) Β· Lamyaa Sadouk Β· EMSI Casablanca*
---
Enter a classroom instruction as a teacher would say it out loud.
VisualStep will decompose it into a structured, ADHD-adapted visual step card.
""")
with gr.Row():
with gr.Column(scale=1):
instruction = gr.Textbox(
label="Teacher's spoken instruction",
placeholder="e.g. Ok everyone, open your books to page 23, read the paragraph quietly, then answer questions 1 to 3...",
lines=4,
)
with gr.Row():
year = gr.Dropdown(
choices=["1", "2", "3", "4", "5", "6"],
value="2",
label="Year group",
)
subject = gr.Dropdown(
choices=["english", "maths", "science", "routine", "transition"],
value="english",
label="Subject",
)
btn = gr.Button("Generate Step Card", variant="primary", size="lg")
with gr.Column(scale=1):
output = gr.Markdown(
label="ADHD-adapted step card",
elem_classes=["card-output"],
value="*Your step card will appear here.*"
)
btn.click(
fn=generate_step_card,
inputs=[instruction, year, subject],
outputs=output,
)
gr.Examples(
examples=EXAMPLES,
inputs=[instruction, year, subject],
label="Try these examples",
)
gr.Markdown("""
---
**About VisualStep:**
This demo accompanies the paper *"VisualStep: A Fine-Tuned Small Language Model for ADHD-Adapted
Visual Instruction Decomposition in Primary School Classrooms"*.
The model was fine-tuned on VisualStep-2K, a dataset of 2,000 spoken classroom instruction–step card pairs.
All inference runs locally β€” no data is sent to external servers.
""")
demo.launch()