"""
IRIS — Intelligent Real-time Inference System for Visual Assistance
Next-Gen Cyber Assistive HUD & Real-time Spatial Navigation
Gradio Application for Hugging Face Spaces (Compatible with ZeroGPU & CPU)
"""
import os
import base64
import tempfile
from pathlib import Path
# Redirect YOLO config/cache to writable directory
os.environ["YOLO_CONFIG_DIR"] = "/tmp/Ultralytics"
# Hugging Face ZeroGPU compatibility
try:
import spaces
except ImportError:
class spaces:
@staticmethod
def GPU(func=None, duration=60):
if func is None:
def decorator(f):
return f
return decorator
return func
import cv2
import numpy as np
from PIL import Image
from ultralytics import YOLO
from gtts import gTTS
import gradio as gr
from priority_engine import PriorityEngine
# ── Config ───────────────────────────────────────────────────────────────────
CONF_THRESHOLD = 0.35
MODEL_NAME = "yolo11n.pt"
# Initialize model & priority engine
model = YOLO(MODEL_NAME)
priority = PriorityEngine()
# Colors for bounding boxes (RGB)
POSITION_COLORS = {
"left": (255, 140, 0), # Orange
"center": (16, 185, 129), # Emerald Green (Directly Ahead)
"right": (59, 130, 246), # Electric Blue
}
def get_position(center_x, frame_width):
if center_x < frame_width / 3:
return "left"
elif center_x < 2 * frame_width / 3:
return "center"
return "right"
def generate_tts_audio(text: str):
"""Generate an MP3 audio file using gTTS."""
if not text or len(text.strip()) == 0:
return None
try:
clean_text = (
text.replace("🔊", "")
.replace("⚠️", "")
.replace("—", ", ")
.replace("–", ", ")
.strip()
)
tts = gTTS(text=clean_text, lang="en", slow=False)
tmp = tempfile.NamedTemporaryFile(suffix=".mp3", delete=False)
tts.save(tmp.name)
return tmp.name
except Exception as e:
print(f"[TTS Error] {e}")
return None
def render_spatial_hud(detections):
"""Render a 3-zone spatial radar HUD and obstacle badges."""
left_items = [d for d in detections if d["position"] == "left"]
center_items = [d for d in detections if d["position"] == "center"]
right_items = [d for d in detections if d["position"] == "right"]
def make_zone(items, zone_name, color):
if not items:
return f"""
"""
top_obj = items[0]["object"].capitalize()
conf = int(items[0]["confidence"] * 100)
extra = f" +{len(items)-1}" if len(items) > 1 else ""
is_warn = zone_name == "CENTER PATH"
val_cls = "val-alert" if is_warn else "val-warn"
return f"""
{top_obj} ({conf}%){extra}
"""
zones_html = f"""
{make_zone(left_items, "LEFT ZONE", "#f97316")}
{make_zone(center_items, "CENTER PATH", "#10b981")}
{make_zone(right_items, "RIGHT ZONE", "#3b82f6")}
"""
if detections:
badges = []
for d in detections:
c = POSITION_COLORS.get(d["position"], (200, 200, 200))
rgb_str = f"rgb({c[0]},{c[1]},{c[2]})"
badges.append(f"""
{d['object'].capitalize()}
{d['position'].upper()}
{int(d['confidence']*100)}%
""")
chips_html = f"""
RADAR DETECTIONS ({len(detections)})
{"".join(badges)}
"""
else:
chips_html = """
RADAR DETECTIONS
✓ Navigation corridor is clear of detected obstacles.
"""
return zones_html + chips_html
@spaces.GPU
def analyze_frame(image):
"""
Run YOLO inference and Priority Engine on an input image.
Returns:
annotated_image, instruction_text, audio_filepath, detections_hud_html
"""
if image is None:
placeholder = """
"""
return None, "STANDBY — Waiting for visual input.", None, placeholder
h, w, _ = image.shape
results = model(image, conf=CONF_THRESHOLD, verbose=False)
detections = []
annotated = image.copy()
for result in results:
for box in result.boxes:
conf = float(box.conf[0])
if conf < CONF_THRESHOLD:
continue
cls_id = int(box.cls[0])
label = model.names[cls_id]
x1, y1, x2, y2 = box.xyxy[0].tolist()
center_x = (x1 + x2) / 2
pos = get_position(center_x, w)
detections.append({
"object": label,
"confidence": round(conf, 2),
"position": pos,
"bbox": [round(x1, 1), round(y1, 1), round(x2, 1), round(y2, 1)],
})
# Draw bounding box and label in RGB
color = POSITION_COLORS.get(pos, (255, 255, 255))
cv2.rectangle(annotated, (int(x1), int(y1)), (int(x2), int(y2)), color, 2)
tag = f"{label.upper()} {int(conf * 100)}%"
(tw, th), _ = cv2.getTextSize(tag, cv2.FONT_HERSHEY_SIMPLEX, 0.55, 2)
ty = max(int(y1) - 8, th + 6)
cv2.rectangle(annotated, (int(x1), ty - th - 4), (int(x1) + tw + 6, ty + 2), color, -1)
cv2.putText(
annotated, tag, (int(x1) + 3, ty - 2),
cv2.FONT_HERSHEY_SIMPLEX, 0.55, (10, 15, 25), 2
)
detections.sort(key=lambda d: d["confidence"], reverse=True)
instruction = priority.pick(detections)
audio_path = generate_tts_audio(instruction)
hud_html = render_spatial_hud(detections)
return annotated, f"🔊 {instruction}", audio_path, hud_html
@spaces.GPU
def analyze_webcam_input(b64_data, fallback_image):
"""Decodes live frame from browser canvas base64 or fallback Gradio snapshot."""
image = None
if b64_data and len(b64_data) > 100:
try:
raw = base64.b64decode(b64_data.split(",")[-1])
arr = np.frombuffer(raw, np.uint8)
bgr = cv2.imdecode(arr, cv2.IMREAD_COLOR)
if bgr is not None:
image = cv2.cvtColor(bgr, cv2.COLOR_BGR2RGB)
except Exception as e:
print(f"[Live frame decode error] {e}")
if image is None and fallback_image is not None:
image = fallback_image
if image is None:
placeholder = """
⚠️ Camera stream inactive. Click "Start Camera" above.
"""
return None, "STANDBY — Please start the camera stream.", None, placeholder
return analyze_frame(image)
@spaces.GPU(duration=120)
def analyze_video(video_path):
"""Sample key frames from an uploaded video and generate navigation guidance."""
if not video_path:
return None, "No video provided", None
cap = cv2.VideoCapture(video_path)
total = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))
fps = cap.get(cv2.CAP_PROP_FPS) or 30
sample_step = max(1, total // 6)
sampled_frames = []
frame_idx = 0
all_instructions = []
while True:
ret, frame = cap.read()
if not ret:
break
if frame_idx % sample_step == 0:
frame_rgb = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)
ann, inst, _, _ = analyze_frame(frame_rgb)
sampled_frames.append(ann)
sec = round(frame_idx / fps, 1)
all_instructions.append(f"⏱️ **{sec}s**: {inst.replace('🔊 ', '')}")
frame_idx += 1
cap.release()
timeline = "\n\n".join(all_instructions)
first_instruction = all_instructions[0].split(": ")[-1] if all_instructions else "Video analysis complete."
overall_tts = generate_tts_audio(first_instruction)
return sampled_frames, timeline, overall_tts
# ── Sequential Live Navigation Controller (Speaks 1st, then scans next) ─────
live_nav_js = """
() => {
if (typeof window.isLiveNavActive === 'undefined') {
window.isLiveNavActive = false;
window.liveNavTimer = null;
}
window.isLiveNavActive = !window.isLiveNavActive;
const btn = document.querySelector("#live_nav_toggle_btn button, #live_nav_toggle_btn");
const status = document.getElementById("live_status_indicator");
function triggerScan() {
if (!window.isLiveNavActive) return;
const scanBtn = document.querySelector("#scan_frame_btn button, #scan_frame_btn");
if (scanBtn && !scanBtn.disabled) {
scanBtn.click();
}
}
// Called as soon as YOLO finishes processing a frame
window.onFrameOutputReceived = function(instructionText) {
if (!instructionText) {
if (window.isLiveNavActive) {
window.liveNavTimer = setTimeout(triggerScan, 1000);
}
return;
}
const clean = instructionText.replace("🔊", "").replace("⚠️", "").replace("STANDBY", "").trim();
if (!clean || clean.includes("Please start") || clean.includes("Waiting for") || clean.includes("Camera stream")) {
if (window.isLiveNavActive) {
window.liveNavTimer = setTimeout(triggerScan, 1200);
}
return;
}
if (!('speechSynthesis' in window)) {
// Fallback if browser doesn't support Web Speech
if (window.isLiveNavActive) {
window.liveNavTimer = setTimeout(triggerScan, 2500);
}
return;
}
// Cancel previous utterance
window.speechSynthesis.cancel();
const u = new SpeechSynthesisUtterance(clean);
u.rate = 0.95;
u.pitch = 1.0;
let hasFinished = false;
function onSpeechComplete() {
if (hasFinished) return;
hasFinished = true;
// Frame output has finished speaking!
// Wait 500ms pause, then trigger the next frame scan:
if (window.isLiveNavActive) {
window.liveNavTimer = setTimeout(triggerScan, 500);
}
}
u.onend = onSpeechComplete;
u.onerror = onSpeechComplete;
// Safety timeout in case browser onend event is dropped
const safeTimeout = Math.max(3500, clean.length * 120);
setTimeout(onSpeechComplete, safeTimeout);
window.speechSynthesis.speak(u);
};
if (window.isLiveNavActive) {
if (btn) {
btn.innerText = "⏹️ Stop Real-Time Navigation";
btn.style.setProperty("background", "linear-gradient(135deg, #ef4444, #dc2626)", "important");
btn.classList.add("active");
}
if (status) {
status.innerHTML = "● ACTIVE — Sequential speech: speaks full alert before scanning next frame.";
}
// Trigger the initial frame scan
triggerScan();
} else {
if (btn) {
btn.innerText = "▶️ Start Real-Time Navigation (Sequential Voice)";
btn.style.setProperty("background", "linear-gradient(135deg, #10b981, #059669)", "important");
btn.classList.remove("active");
}
if (status) {
status.innerHTML = "● STANDBY — Click above to begin continuous scanning.";
}
if (window.liveNavTimer) {
clearTimeout(window.liveNavTimer);
window.liveNavTimer = null;
}
if ('speechSynthesis' in window) {
window.speechSynthesis.cancel();
}
}
}
"""
# Extract current live frame from HTML5 video element with no shutter lag
js_extract_video = """
(b64, fallback_img) => {
const video = document.querySelector("#webcam_viewport video");
if (video && video.videoWidth > 0) {
const canvas = document.createElement("canvas");
canvas.width = video.videoWidth;
canvas.height = video.videoHeight;
const ctx = canvas.getContext("2d");
ctx.drawImage(video, 0, 0, canvas.width, canvas.height);
return [canvas.toDataURL("image/jpeg", 0.85), null];
}
return [b64, fallback_img];
}
"""
# ── Ultra-Professional Dark Theme CSS ────────────────────────────────────────
custom_css = """
"""
# ── Build Gradio Interface ───────────────────────────────────────────────────
with gr.Blocks(title="IRIS — Intelligent Visual Copilot") as demo:
gr.HTML(custom_css)
# Top Brand Bar
gr.HTML("""
IRIS VISUAL COPILOT
Real-Time Spatial Obstacle Detection & Voice Guidance
SYSTEM ONLINE
YOLO11-NANO
3-ZONE RADAR
""")
with gr.Tabs():
# ── TAB 1: Real-Time Live Vision ─────────────────────────────────────
with gr.TabItem("📷 Live Visual Navigation", id="tab_live"):
with gr.Row():
with gr.Column(scale=5):
webcam_input = gr.Image(
sources=["webcam"],
type="numpy",
label="Live Optical Sensor",
elem_id="webcam_viewport",
)
webcam_b64 = gr.Textbox(visible=False, elem_id="webcam_b64_buffer")
# Native Gradio button for live navigation toggle
live_nav_btn = gr.Button(
"▶️ Start Real-Time Navigation (Sequential Voice)",
elem_id="live_nav_toggle_btn",
elem_classes=["live-stream-btn"],
variant="primary",
)
gr.HTML("""
● STANDBY — Click above to begin continuous scanning.
""")
webcam_snap_btn = gr.Button(
"⚡ Instant Scan (Current Frame)",
elem_id="scan_frame_btn",
elem_classes=["btn-primary-action"],
variant="secondary",
)
with gr.Column(scale=6):
webcam_instruction = gr.Textbox(
label="📢 Priority Voice Navigation Alert",
interactive=False,
elem_id="webcam_instruction",
elem_classes=["instruction-box"],
value="STANDBY — Waiting for visual input.",
)
webcam_audio = gr.Audio(
label="🔊 Audio Recording (Optional)",
autoplay=False,
)
webcam_hud = gr.HTML(
label="Spatial Radar Breakdown",
value="""
""",
)
webcam_output = gr.Image(
label="Annotated Spatial View",
)
# Wire up the instant scan button (extracts live frame via JS)
# When inference completes, .then() triggers sequential voice & next scan
webcam_snap_btn.click(
fn=analyze_webcam_input,
inputs=[webcam_b64, webcam_input],
outputs=[webcam_output, webcam_instruction, webcam_audio, webcam_hud],
js=js_extract_video,
).then(
fn=None,
inputs=[webcam_instruction],
js="""(inst) => { if (window.onFrameOutputReceived) { window.onFrameOutputReceived(inst); } }""",
)
# Wire up the live navigation button to toggle the continuous loop
live_nav_btn.click(
fn=None,
js=live_nav_js,
)
# ── TAB 2: Image Diagnostic ──────────────────────────────────────────
with gr.TabItem("🖼️ Photo Inspection", id="tab_photo"):
with gr.Row():
with gr.Column(scale=5):
upload_input = gr.Image(
sources=["upload"],
type="numpy",
label="Upload Scene Photograph",
)
upload_btn = gr.Button(
"🔍 Analyze Scene Obstacles",
elem_classes=["btn-primary-action"],
variant="primary",
)
with gr.Column(scale=6):
upload_instruction = gr.Textbox(
label="📢 Spoken Navigation Instruction",
interactive=False,
elem_classes=["instruction-box"],
)
upload_audio = gr.Audio(
label="🔊 Audio Instruction",
autoplay=True,
)
upload_hud = gr.HTML(
label="Spatial Radar Breakdown",
)
upload_output = gr.Image(
label="Annotated Spatial View",
)
upload_btn.click(
fn=analyze_frame,
inputs=[upload_input],
outputs=[upload_output, upload_instruction, upload_audio, upload_hud],
)
# ── TAB 3: Video Walkthrough ─────────────────────────────────────────
with gr.TabItem("🎥 Video Walkthrough", id="tab_video"):
with gr.Row():
with gr.Column(scale=5):
video_input = gr.Video(
sources=["upload"],
label="Upload Navigation Footage (.mp4, .mov)",
)
video_btn = gr.Button(
"🎬 Analyze Navigation Corridor",
elem_classes=["btn-primary-action"],
variant="primary",
)
with gr.Column(scale=6):
video_audio = gr.Audio(
label="🔊 Spoken Overview",
autoplay=True,
)
video_timeline = gr.Markdown(
label="Chronological Timeline Guidance",
)
video_gallery = gr.Gallery(
label="Corridor Keyframe Snapshots",
columns=2,
)
video_btn.click(
fn=analyze_video,
inputs=[video_input],
outputs=[video_gallery, video_timeline, video_audio],
)
# ── TAB 4: System Architecture ───────────────────────────────────────
with gr.TabItem("ℹ️ System Architecture", id="tab_about"):
gr.HTML("""
🧭 IRIS Navigation Engine Principles
1. 3-Zone Spatial Radar
The sensor field is dynamically partitioned into LEFT (33%) , CENTER PATH (33%) , and RIGHT (33%) . Obstacles located directly in the center path receive critical priority.
2. Zero-Cognitive-Overload Priority
Instead of overwhelming a visually impaired person with a list of 10 items, IRIS selects exactly ONE urgent, actionable instruction (e.g. "Caution! Person directly ahead" or "Watch out — chair ahead. Step around it" ).
3. Dual-Stream Voice Guidance
Low-latency browser speech synthesis provides instant voice cues on device, complemented by server-side synthesized audio with automatic playback.
""")
if __name__ == "__main__":
demo.launch()