"""Audio Insight: upload recordings, get an analysis report and cross-file patterns. Hugging Face Space entry point (Gradio). Set the Space secret GEMINI_API_KEY (or GEMINI_API_KEYS with several comma-separated keys) to enable the AI parts. """ from __future__ import annotations import html import os import shutil import tempfile import threading import time import gradio as gr import numpy as np import config import pipeline import report from ai_analysis import fmt_time # Free Hugging Face accounts can only run Gradio Spaces on ZeroGPU hardware, which # refuses to start an app without at least one @spaces.GPU function. This app needs # no GPU; the placeholder below is never called and is skipped outside Spaces. try: import spaces @spaces.GPU def _zerogpu_placeholder() -> None: return None except Exception: # not on Spaces / package absent pass OUT_ROOT = os.path.join(tempfile.gettempdir(), "audio_insight_reports") KEEP_OUTPUTS_S = 2 * 3600 def _prewarm() -> None: """Import heavy libraries and compile numba kernels before the first user arrives.""" try: import acoustics import charts ac = acoustics.analyze((np.random.default_rng(0).standard_normal(16000 * 3) * 0.05).astype(np.float32)) charts.spectrogram_png(ac.mel_db, ac.duration_s) except Exception: pass threading.Thread(target=_prewarm, daemon=True).start() def _clean_old_outputs() -> None: if not os.path.isdir(OUT_ROOT): return now = time.time() for entry in os.scandir(OUT_ROOT): try: if now - entry.stat().st_mtime > KEEP_OUTPUTS_S: shutil.rmtree(entry.path, ignore_errors=True) except OSError: pass def _paths(files) -> list[str]: out = [] for f in files or []: p = f if isinstance(f, str) else getattr(f, "name", None) or getattr(f, "path", None) if p: out.append(p) return out def _summary_md(run: dict, outs: dict) -> str: ok = [r for r in run["files"] if not r.get("error")] total = sum(r["acoustics"].duration_s for r in ok) models = ", ".join(run.get("usage", {}).get("models", [])) or "measured analysis only (no Gemini)" lines = [f"**Analysed {len(ok)} of {len(run['files'])} file(s)** · {fmt_time(total)} of audio · " f"{run['elapsed_s']:.0f} s · {models}"] for r in run["files"]: if r.get("error"): lines.append(f"- ✕ **{r['id']}** {r['name']}: {r['error']}") elif r.get("ai_error"): lines.append(f"- ! **{r['id']}** {r['name']}: {r['ai_error']}") if run.get("synthesis_error"): lines.append(f"- ! Cross-file synthesis: {run['synthesis_error']}") for w in run.get("warnings", []): lines.append(f"- ! {w}") lines.append("\nThe full report is below; download it (HTML, opens in any browser and prints to PDF) " "or the ZIP with the report, JSON data, metrics CSV and transcripts (TXT/SRT).") return "\n".join(lines) def analyse(files, context, questions, use_ai, deep, api_key, progress=gr.Progress()): paths = _paths(files) if not paths: raise gr.Error("Upload at least one audio file.") _clean_old_outputs() names = [os.path.basename(p) for p in paths] # The pipeline reports progress from worker threads, where gr.Progress has no # effect; it records the latest state and this request thread forwards it. state = {"frac": 0.0, "msg": "Starting"} lock = threading.Lock() result: dict = {} def cb(frac: float, msg: str) -> None: with lock: state["frac"], state["msg"] = min(max(frac, 0.0), 1.0), msg def work() -> None: try: result["run"] = pipeline.run(paths, names=names, context=context or "", questions=questions or "", use_ai=bool(use_ai), deep=bool(deep), api_key=(api_key or "").strip() or None, progress=cb) except BaseException as exc: # re-raised on the request thread below result["error"] = exc worker = threading.Thread(target=work, daemon=True) worker.start() shown = None while worker.is_alive(): worker.join(0.4) with lock: now = (state["frac"], state["msg"]) if now != shown: progress(now[0], desc=now[1]) shown = now if "error" in result: exc = result["error"] if isinstance(exc, ValueError): raise gr.Error(str(exc)) raise exc run = result["run"] os.makedirs(OUT_ROOT, exist_ok=True) out_dir = tempfile.mkdtemp(prefix="run-", dir=OUT_ROOT) outs = report.write_outputs(run, out_dir) frame = (f'') return _summary_md(run, outs), frame, [outs["html"], outs["zip"], outs["csv"], outs["json"]] def _key_status() -> str: n = len(config.api_keys()) if n: return f"Gemini is configured on this Space ({n} key{'s' if n > 1 else ''})." return ("No Gemini key is configured on this Space: paste your own key under *Options*, " "or run the measured analysis only.") INTRO = f""" # Audio Insight Upload one or more recordings (interviews, meetings, lectures, calls, podcasts, music, field recordings). You get a report for each file (transcript, speakers, themes, sentiment, audio quality, loudness, pauses, pitch) and, with two or more files, the **patterns across them**: shared themes, differences, trends over time and acoustic / vocabulary similarity. {_key_status()} Audio is analysed in memory and sent to Google Gemini for the AI parts; nothing is kept after the session. Use a paid-tier Gemini key for confidential recordings. Limits: {config.MAX_FILES} files, {config.MAX_MINUTES_PER_FILE} min per file, {config.MAX_TOTAL_MINUTES} min per run. """ CSS = """ html,body{background:#ffffff!important;color-scheme:light} .report-frame{width:100%;height:900px;border:1px solid #dcdcdc;border-radius:6px;background:#ffffff} footer{display:none!important} """ # Plain, professional look: white page, black text, grey lines, black primary button. FONT = ["-apple-system", "BlinkMacSystemFont", "Segoe UI", "Helvetica Neue", "Arial", "sans-serif"] THEME = gr.themes.Base(primary_hue=gr.themes.colors.neutral, secondary_hue=gr.themes.colors.neutral, neutral_hue=gr.themes.colors.neutral, radius_size=gr.themes.sizes.radius_sm, font=FONT).set( body_background_fill="#ffffff", body_text_color="#111111", body_text_color_subdued="#555555", background_fill_primary="#ffffff", background_fill_secondary="#fafafa", block_background_fill="#ffffff", block_border_color="#dcdcdc", border_color_primary="#dcdcdc", block_label_background_fill="#ffffff", block_label_text_color="#111111", block_title_text_color="#111111", input_background_fill="#ffffff", link_text_color="#111111", color_accent="#111111", color_accent_soft="#eeeeee", loader_color="#111111", slider_color="#111111", checkbox_background_color_selected="#111111", checkbox_border_color_selected="#111111", button_primary_background_fill="#111111", button_primary_background_fill_hover="#333333", button_primary_text_color="#ffffff", button_primary_border_color="#111111", ) # Always render in light mode (white background), even when the visitor's system is dark. FORCE_LIGHT = """ () => { const url = new URL(window.location.href); if (url.searchParams.get('__theme') !== 'light') { url.searchParams.set('__theme', 'light'); window.location.replace(url.href); } } """ with gr.Blocks(title="Audio Insight", delete_cache=(3600, 7200)) as demo: gr.Markdown(INTRO) with gr.Row(equal_height=False): with gr.Column(scale=1): files = gr.File(label="Audio files", file_count="multiple", file_types=["audio", "video"] + config.AUDIO_EXTENSIONS, height=220) with gr.Column(scale=1): context = gr.Textbox( label="What are these recordings? (optional)", lines=3, max_lines=8, placeholder="e.g. Research interviews with nurses about AI documentation tools. " "Names and technical terms help the transcript.") questions = gr.Textbox( label="Questions to answer across the recordings (optional, one per line)", lines=3, max_lines=10, placeholder="What pain points do participants describe?\nHow do they feel about AI?") with gr.Accordion("Options", open=False): with gr.Row(): use_ai = gr.Checkbox(value=True, label="Use Gemini (transcript, themes, sentiment, cross-file patterns)") deep = gr.Checkbox(value=False, label="Deeper reasoning (Gemini Pro writes the analysis; slower)") api_key = gr.Textbox(label="Gemini API key (optional)", type="password", placeholder="Leave empty to use the Space's key. Your key is used for this run only and never stored.") run_btn = gr.Button("Analyse", variant="primary", size="lg") status = gr.Markdown() report_view = gr.HTML() downloads = gr.File(label="Downloads", file_count="multiple", interactive=False) run_btn.click(analyse, [files, context, questions, use_ai, deep, api_key], [status, report_view, downloads], concurrency_limit=2, show_progress="full") if __name__ == "__main__": demo.queue(default_concurrency_limit=2, max_size=20) demo.launch(css=CSS, theme=THEME, js=FORCE_LIGHT, max_file_size=f"{config.MAX_UPLOAD_MB}mb", allowed_paths=[OUT_ROOT])