""" IRIS — Intelligent Real-time Inference System for Visual Assistance Next-Gen Cyber Assistive HUD & Real-time Spatial Navigation Gradio Application for Hugging Face Spaces (Compatible with ZeroGPU & CPU) """ import os import base64 import tempfile from pathlib import Path # Redirect YOLO config/cache to writable directory os.environ["YOLO_CONFIG_DIR"] = "/tmp/Ultralytics" # Hugging Face ZeroGPU compatibility try: import spaces except ImportError: class spaces: @staticmethod def GPU(func=None, duration=60): if func is None: def decorator(f): return f return decorator return func import cv2 import numpy as np from PIL import Image from ultralytics import YOLO from gtts import gTTS import gradio as gr from priority_engine import PriorityEngine # ── Config ─────────────────────────────────────────────────────────────────── CONF_THRESHOLD = 0.35 MODEL_NAME = "yolo11n.pt" # Initialize model & priority engine model = YOLO(MODEL_NAME) priority = PriorityEngine() # Colors for bounding boxes (RGB) POSITION_COLORS = { "left": (255, 140, 0), # Orange "center": (16, 185, 129), # Emerald Green (Directly Ahead) "right": (59, 130, 246), # Electric Blue } def get_position(center_x, frame_width): if center_x < frame_width / 3: return "left" elif center_x < 2 * frame_width / 3: return "center" return "right" def generate_tts_audio(text: str): """Generate an MP3 audio file using gTTS.""" if not text or len(text.strip()) == 0: return None try: clean_text = ( text.replace("🔊", "") .replace("⚠️", "") .replace("—", ", ") .replace("–", ", ") .strip() ) tts = gTTS(text=clean_text, lang="en", slow=False) tmp = tempfile.NamedTemporaryFile(suffix=".mp3", delete=False) tts.save(tmp.name) return tmp.name except Exception as e: print(f"[TTS Error] {e}") return None def render_spatial_hud(detections): """Render a 3-zone spatial radar HUD and obstacle badges.""" left_items = [d for d in detections if d["position"] == "left"] center_items = [d for d in detections if d["position"] == "center"] right_items = [d for d in detections if d["position"] == "right"] def make_zone(items, zone_name, color): if not items: return f"""
{zone_name}
CLEAR
""" top_obj = items[0]["object"].capitalize() conf = int(items[0]["confidence"] * 100) extra = f" +{len(items)-1}" if len(items) > 1 else "" is_warn = zone_name == "CENTER PATH" val_cls = "val-alert" if is_warn else "val-warn" return f"""
{zone_name}
{top_obj} ({conf}%){extra}
""" zones_html = f"""
{make_zone(left_items, "LEFT ZONE", "#f97316")} {make_zone(center_items, "CENTER PATH", "#10b981")} {make_zone(right_items, "RIGHT ZONE", "#3b82f6")}
""" if detections: badges = [] for d in detections: c = POSITION_COLORS.get(d["position"], (200, 200, 200)) rgb_str = f"rgb({c[0]},{c[1]},{c[2]})" badges.append(f"""
{d['object'].capitalize()} {d['position'].upper()} {int(d['confidence']*100)}%
""") chips_html = f"""
RADAR DETECTIONS ({len(detections)})
{"".join(badges)}
""" else: chips_html = """
RADAR DETECTIONS
✓ Navigation corridor is clear of detected obstacles.
""" return zones_html + chips_html @spaces.GPU def analyze_frame(image): """ Run YOLO inference and Priority Engine on an input image. Returns: annotated_image, instruction_text, audio_filepath, detections_hud_html """ if image is None: placeholder = """
STANDBY
""" return None, "STANDBY — Waiting for visual input.", None, placeholder h, w, _ = image.shape results = model(image, conf=CONF_THRESHOLD, verbose=False) detections = [] annotated = image.copy() for result in results: for box in result.boxes: conf = float(box.conf[0]) if conf < CONF_THRESHOLD: continue cls_id = int(box.cls[0]) label = model.names[cls_id] x1, y1, x2, y2 = box.xyxy[0].tolist() center_x = (x1 + x2) / 2 pos = get_position(center_x, w) detections.append({ "object": label, "confidence": round(conf, 2), "position": pos, "bbox": [round(x1, 1), round(y1, 1), round(x2, 1), round(y2, 1)], }) # Draw bounding box and label in RGB color = POSITION_COLORS.get(pos, (255, 255, 255)) cv2.rectangle(annotated, (int(x1), int(y1)), (int(x2), int(y2)), color, 2) tag = f"{label.upper()} {int(conf * 100)}%" (tw, th), _ = cv2.getTextSize(tag, cv2.FONT_HERSHEY_SIMPLEX, 0.55, 2) ty = max(int(y1) - 8, th + 6) cv2.rectangle(annotated, (int(x1), ty - th - 4), (int(x1) + tw + 6, ty + 2), color, -1) cv2.putText( annotated, tag, (int(x1) + 3, ty - 2), cv2.FONT_HERSHEY_SIMPLEX, 0.55, (10, 15, 25), 2 ) detections.sort(key=lambda d: d["confidence"], reverse=True) instruction = priority.pick(detections) audio_path = generate_tts_audio(instruction) hud_html = render_spatial_hud(detections) return annotated, f"🔊 {instruction}", audio_path, hud_html @spaces.GPU def analyze_webcam_input(b64_data, fallback_image): """Decodes live frame from browser canvas base64 or fallback Gradio snapshot.""" image = None if b64_data and len(b64_data) > 100: try: raw = base64.b64decode(b64_data.split(",")[-1]) arr = np.frombuffer(raw, np.uint8) bgr = cv2.imdecode(arr, cv2.IMREAD_COLOR) if bgr is not None: image = cv2.cvtColor(bgr, cv2.COLOR_BGR2RGB) except Exception as e: print(f"[Live frame decode error] {e}") if image is None and fallback_image is not None: image = fallback_image if image is None: placeholder = """
⚠️ Camera stream inactive. Click "Start Camera" above.
""" return None, "STANDBY — Please start the camera stream.", None, placeholder return analyze_frame(image) @spaces.GPU(duration=120) def analyze_video(video_path): """Sample key frames from an uploaded video and generate navigation guidance.""" if not video_path: return None, "No video provided", None cap = cv2.VideoCapture(video_path) total = int(cap.get(cv2.CAP_PROP_FRAME_COUNT)) fps = cap.get(cv2.CAP_PROP_FPS) or 30 sample_step = max(1, total // 6) sampled_frames = [] frame_idx = 0 all_instructions = [] while True: ret, frame = cap.read() if not ret: break if frame_idx % sample_step == 0: frame_rgb = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB) ann, inst, _, _ = analyze_frame(frame_rgb) sampled_frames.append(ann) sec = round(frame_idx / fps, 1) all_instructions.append(f"⏱️ **{sec}s**: {inst.replace('🔊 ', '')}") frame_idx += 1 cap.release() timeline = "\n\n".join(all_instructions) first_instruction = all_instructions[0].split(": ")[-1] if all_instructions else "Video analysis complete." overall_tts = generate_tts_audio(first_instruction) return sampled_frames, timeline, overall_tts # ── Sequential Live Navigation Controller (Speaks 1st, then scans next) ───── live_nav_js = """ () => { if (typeof window.isLiveNavActive === 'undefined') { window.isLiveNavActive = false; window.liveNavTimer = null; } window.isLiveNavActive = !window.isLiveNavActive; const btn = document.querySelector("#live_nav_toggle_btn button, #live_nav_toggle_btn"); const status = document.getElementById("live_status_indicator"); function triggerScan() { if (!window.isLiveNavActive) return; const scanBtn = document.querySelector("#scan_frame_btn button, #scan_frame_btn"); if (scanBtn && !scanBtn.disabled) { scanBtn.click(); } } // Called as soon as YOLO finishes processing a frame window.onFrameOutputReceived = function(instructionText) { if (!instructionText) { if (window.isLiveNavActive) { window.liveNavTimer = setTimeout(triggerScan, 1000); } return; } const clean = instructionText.replace("🔊", "").replace("⚠️", "").replace("STANDBY", "").trim(); if (!clean || clean.includes("Please start") || clean.includes("Waiting for") || clean.includes("Camera stream")) { if (window.isLiveNavActive) { window.liveNavTimer = setTimeout(triggerScan, 1200); } return; } if (!('speechSynthesis' in window)) { // Fallback if browser doesn't support Web Speech if (window.isLiveNavActive) { window.liveNavTimer = setTimeout(triggerScan, 2500); } return; } // Cancel previous utterance window.speechSynthesis.cancel(); const u = new SpeechSynthesisUtterance(clean); u.rate = 0.95; u.pitch = 1.0; let hasFinished = false; function onSpeechComplete() { if (hasFinished) return; hasFinished = true; // Frame output has finished speaking! // Wait 500ms pause, then trigger the next frame scan: if (window.isLiveNavActive) { window.liveNavTimer = setTimeout(triggerScan, 500); } } u.onend = onSpeechComplete; u.onerror = onSpeechComplete; // Safety timeout in case browser onend event is dropped const safeTimeout = Math.max(3500, clean.length * 120); setTimeout(onSpeechComplete, safeTimeout); window.speechSynthesis.speak(u); }; if (window.isLiveNavActive) { if (btn) { btn.innerText = "⏹️ Stop Real-Time Navigation"; btn.style.setProperty("background", "linear-gradient(135deg, #ef4444, #dc2626)", "important"); btn.classList.add("active"); } if (status) { status.innerHTML = "● ACTIVE — Sequential speech: speaks full alert before scanning next frame."; } // Trigger the initial frame scan triggerScan(); } else { if (btn) { btn.innerText = "▶️ Start Real-Time Navigation (Sequential Voice)"; btn.style.setProperty("background", "linear-gradient(135deg, #10b981, #059669)", "important"); btn.classList.remove("active"); } if (status) { status.innerHTML = "● STANDBY — Click above to begin continuous scanning."; } if (window.liveNavTimer) { clearTimeout(window.liveNavTimer); window.liveNavTimer = null; } if ('speechSynthesis' in window) { window.speechSynthesis.cancel(); } } } """ # Extract current live frame from HTML5 video element with no shutter lag js_extract_video = """ (b64, fallback_img) => { const video = document.querySelector("#webcam_viewport video"); if (video && video.videoWidth > 0) { const canvas = document.createElement("canvas"); canvas.width = video.videoWidth; canvas.height = video.videoHeight; const ctx = canvas.getContext("2d"); ctx.drawImage(video, 0, 0, canvas.width, canvas.height); return [canvas.toDataURL("image/jpeg", 0.85), null]; } return [b64, fallback_img]; } """ # ── Ultra-Professional Dark Theme CSS ──────────────────────────────────────── custom_css = """ """ # ── Build Gradio Interface ─────────────────────────────────────────────────── with gr.Blocks(title="IRIS — Intelligent Visual Copilot") as demo: gr.HTML(custom_css) # Top Brand Bar gr.HTML("""

IRIS VISUAL COPILOT

Real-Time Spatial Obstacle Detection & Voice Guidance

SYSTEM ONLINE
YOLO11-NANO
3-ZONE RADAR
""") with gr.Tabs(): # ── TAB 1: Real-Time Live Vision ───────────────────────────────────── with gr.TabItem("📷 Live Visual Navigation", id="tab_live"): with gr.Row(): with gr.Column(scale=5): webcam_input = gr.Image( sources=["webcam"], type="numpy", label="Live Optical Sensor", elem_id="webcam_viewport", ) webcam_b64 = gr.Textbox(visible=False, elem_id="webcam_b64_buffer") # Native Gradio button for live navigation toggle live_nav_btn = gr.Button( "▶️ Start Real-Time Navigation (Sequential Voice)", elem_id="live_nav_toggle_btn", elem_classes=["live-stream-btn"], variant="primary", ) gr.HTML("""

● STANDBY — Click above to begin continuous scanning.

""") webcam_snap_btn = gr.Button( "⚡ Instant Scan (Current Frame)", elem_id="scan_frame_btn", elem_classes=["btn-primary-action"], variant="secondary", ) with gr.Column(scale=6): webcam_instruction = gr.Textbox( label="📢 Priority Voice Navigation Alert", interactive=False, elem_id="webcam_instruction", elem_classes=["instruction-box"], value="STANDBY — Waiting for visual input.", ) webcam_audio = gr.Audio( label="🔊 Audio Recording (Optional)", autoplay=False, ) webcam_hud = gr.HTML( label="Spatial Radar Breakdown", value="""
STANDBY
""", ) webcam_output = gr.Image( label="Annotated Spatial View", ) # Wire up the instant scan button (extracts live frame via JS) # When inference completes, .then() triggers sequential voice & next scan webcam_snap_btn.click( fn=analyze_webcam_input, inputs=[webcam_b64, webcam_input], outputs=[webcam_output, webcam_instruction, webcam_audio, webcam_hud], js=js_extract_video, ).then( fn=None, inputs=[webcam_instruction], js="""(inst) => { if (window.onFrameOutputReceived) { window.onFrameOutputReceived(inst); } }""", ) # Wire up the live navigation button to toggle the continuous loop live_nav_btn.click( fn=None, js=live_nav_js, ) # ── TAB 2: Image Diagnostic ────────────────────────────────────────── with gr.TabItem("🖼️ Photo Inspection", id="tab_photo"): with gr.Row(): with gr.Column(scale=5): upload_input = gr.Image( sources=["upload"], type="numpy", label="Upload Scene Photograph", ) upload_btn = gr.Button( "🔍 Analyze Scene Obstacles", elem_classes=["btn-primary-action"], variant="primary", ) with gr.Column(scale=6): upload_instruction = gr.Textbox( label="📢 Spoken Navigation Instruction", interactive=False, elem_classes=["instruction-box"], ) upload_audio = gr.Audio( label="🔊 Audio Instruction", autoplay=True, ) upload_hud = gr.HTML( label="Spatial Radar Breakdown", ) upload_output = gr.Image( label="Annotated Spatial View", ) upload_btn.click( fn=analyze_frame, inputs=[upload_input], outputs=[upload_output, upload_instruction, upload_audio, upload_hud], ) # ── TAB 3: Video Walkthrough ───────────────────────────────────────── with gr.TabItem("🎥 Video Walkthrough", id="tab_video"): with gr.Row(): with gr.Column(scale=5): video_input = gr.Video( sources=["upload"], label="Upload Navigation Footage (.mp4, .mov)", ) video_btn = gr.Button( "🎬 Analyze Navigation Corridor", elem_classes=["btn-primary-action"], variant="primary", ) with gr.Column(scale=6): video_audio = gr.Audio( label="🔊 Spoken Overview", autoplay=True, ) video_timeline = gr.Markdown( label="Chronological Timeline Guidance", ) video_gallery = gr.Gallery( label="Corridor Keyframe Snapshots", columns=2, ) video_btn.click( fn=analyze_video, inputs=[video_input], outputs=[video_gallery, video_timeline, video_audio], ) # ── TAB 4: System Architecture ─────────────────────────────────────── with gr.TabItem("ℹ️ System Architecture", id="tab_about"): gr.HTML("""

🧭 IRIS Navigation Engine Principles

1. 3-Zone Spatial Radar

The sensor field is dynamically partitioned into LEFT (33%), CENTER PATH (33%), and RIGHT (33%). Obstacles located directly in the center path receive critical priority.

2. Zero-Cognitive-Overload Priority

Instead of overwhelming a visually impaired person with a list of 10 items, IRIS selects exactly ONE urgent, actionable instruction (e.g. "Caution! Person directly ahead" or "Watch out — chair ahead. Step around it").

3. Dual-Stream Voice Guidance

Low-latency browser speech synthesis provides instant voice cues on device, complemented by server-side synthesized audio with automatic playback.

""") if __name__ == "__main__": demo.launch()