Sam-Just / app.py
stevafernandes's picture
Upload 13 files
2ccc92d verified
Raw History Blame Contribute Delete
9.91 kB
"""Audio Insight: upload recordings, get an analysis report and cross-file patterns.
Hugging Face Space entry point (Gradio). Set the Space secret GEMINI_API_KEY
(or GEMINI_API_KEYS with several comma-separated keys) to enable the AI parts.
"""
from __future__ import annotations
import html
import os
import shutil
import tempfile
import threading
import time
import gradio as gr
import numpy as np
import config
import pipeline
import report
from ai_analysis import fmt_time
# Free Hugging Face accounts can only run Gradio Spaces on ZeroGPU hardware, which
# refuses to start an app without at least one @spaces.GPU function. This app needs
# no GPU; the placeholder below is never called and is skipped outside Spaces.
try:
import spaces
@spaces.GPU
def _zerogpu_placeholder() -> None:
return None
except Exception: # not on Spaces / package absent
pass
OUT_ROOT = os.path.join(tempfile.gettempdir(), "audio_insight_reports")
KEEP_OUTPUTS_S = 2 * 3600
def _prewarm() -> None:
"""Import heavy libraries and compile numba kernels before the first user arrives."""
try:
import acoustics
import charts
ac = acoustics.analyze((np.random.default_rng(0).standard_normal(16000 * 3) * 0.05).astype(np.float32))
charts.spectrogram_png(ac.mel_db, ac.duration_s)
except Exception:
pass
threading.Thread(target=_prewarm, daemon=True).start()
def _clean_old_outputs() -> None:
if not os.path.isdir(OUT_ROOT):
return
now = time.time()
for entry in os.scandir(OUT_ROOT):
try:
if now - entry.stat().st_mtime > KEEP_OUTPUTS_S:
shutil.rmtree(entry.path, ignore_errors=True)
except OSError:
pass
def _paths(files) -> list[str]:
out = []
for f in files or []:
p = f if isinstance(f, str) else getattr(f, "name", None) or getattr(f, "path", None)
if p:
out.append(p)
return out
def _summary_md(run: dict, outs: dict) -> str:
ok = [r for r in run["files"] if not r.get("error")]
total = sum(r["acoustics"].duration_s for r in ok)
models = ", ".join(run.get("usage", {}).get("models", [])) or "measured analysis only (no Gemini)"
lines = [f"**Analysed {len(ok)} of {len(run['files'])} file(s)** · {fmt_time(total)} of audio · "
f"{run['elapsed_s']:.0f} s · {models}"]
for r in run["files"]:
if r.get("error"):
lines.append(f"- ✕ **{r['id']}** {r['name']}: {r['error']}")
elif r.get("ai_error"):
lines.append(f"- ! **{r['id']}** {r['name']}: {r['ai_error']}")
if run.get("synthesis_error"):
lines.append(f"- ! Cross-file synthesis: {run['synthesis_error']}")
for w in run.get("warnings", []):
lines.append(f"- ! {w}")
lines.append("\nThe full report is below; download it (HTML, opens in any browser and prints to PDF) "
"or the ZIP with the report, JSON data, metrics CSV and transcripts (TXT/SRT).")
return "\n".join(lines)
def analyse(files, context, questions, use_ai, deep, api_key, progress=gr.Progress()):
paths = _paths(files)
if not paths:
raise gr.Error("Upload at least one audio file.")
_clean_old_outputs()
names = [os.path.basename(p) for p in paths]
# The pipeline reports progress from worker threads, where gr.Progress has no
# effect; it records the latest state and this request thread forwards it.
state = {"frac": 0.0, "msg": "Starting"}
lock = threading.Lock()
result: dict = {}
def cb(frac: float, msg: str) -> None:
with lock:
state["frac"], state["msg"] = min(max(frac, 0.0), 1.0), msg
def work() -> None:
try:
result["run"] = pipeline.run(paths, names=names, context=context or "", questions=questions or "",
use_ai=bool(use_ai), deep=bool(deep),
api_key=(api_key or "").strip() or None, progress=cb)
except BaseException as exc: # re-raised on the request thread below
result["error"] = exc
worker = threading.Thread(target=work, daemon=True)
worker.start()
shown = None
while worker.is_alive():
worker.join(0.4)
with lock:
now = (state["frac"], state["msg"])
if now != shown:
progress(now[0], desc=now[1])
shown = now
if "error" in result:
exc = result["error"]
if isinstance(exc, ValueError):
raise gr.Error(str(exc))
raise exc
run = result["run"]
os.makedirs(OUT_ROOT, exist_ok=True)
out_dir = tempfile.mkdtemp(prefix="run-", dir=OUT_ROOT)
outs = report.write_outputs(run, out_dir)
frame = (f'<iframe class="report-frame" title="Analysis report" '
f'sandbox="allow-scripts allow-popups allow-popups-to-escape-sandbox" '
f'srcdoc="{html.escape(outs["html_text"], quote=True)}"></iframe>')
return _summary_md(run, outs), frame, [outs["html"], outs["zip"], outs["csv"], outs["json"]]
def _key_status() -> str:
n = len(config.api_keys())
if n:
return f"Gemini is configured on this Space ({n} key{'s' if n > 1 else ''})."
return ("No Gemini key is configured on this Space: paste your own key under *Options*, "
"or run the measured analysis only.")
INTRO = f"""
# Audio Insight
Upload one or more recordings (interviews, meetings, lectures, calls, podcasts, music, field recordings).
You get a report for each file (transcript, speakers, themes, sentiment, audio quality, loudness, pauses, pitch)
and, with two or more files, the **patterns across them**: shared themes, differences, trends over time and
acoustic / vocabulary similarity.
<small>{_key_status()} Audio is analysed in memory and sent to Google Gemini for the AI parts; nothing is kept
after the session. Use a paid-tier Gemini key for confidential recordings.
Limits: {config.MAX_FILES} files, {config.MAX_MINUTES_PER_FILE} min per file, {config.MAX_TOTAL_MINUTES} min per run.</small>
"""
CSS = """
html,body{background:#ffffff!important;color-scheme:light}
.report-frame{width:100%;height:900px;border:1px solid #dcdcdc;border-radius:6px;background:#ffffff}
footer{display:none!important}
"""
# Plain, professional look: white page, black text, grey lines, black primary button.
FONT = ["-apple-system", "BlinkMacSystemFont", "Segoe UI", "Helvetica Neue", "Arial", "sans-serif"]
THEME = gr.themes.Base(primary_hue=gr.themes.colors.neutral, secondary_hue=gr.themes.colors.neutral,
neutral_hue=gr.themes.colors.neutral, radius_size=gr.themes.sizes.radius_sm,
font=FONT).set(
body_background_fill="#ffffff", body_text_color="#111111", body_text_color_subdued="#555555",
background_fill_primary="#ffffff", background_fill_secondary="#fafafa",
block_background_fill="#ffffff", block_border_color="#dcdcdc", border_color_primary="#dcdcdc",
block_label_background_fill="#ffffff", block_label_text_color="#111111", block_title_text_color="#111111",
input_background_fill="#ffffff", link_text_color="#111111",
color_accent="#111111", color_accent_soft="#eeeeee", loader_color="#111111", slider_color="#111111",
checkbox_background_color_selected="#111111", checkbox_border_color_selected="#111111",
button_primary_background_fill="#111111", button_primary_background_fill_hover="#333333",
button_primary_text_color="#ffffff", button_primary_border_color="#111111",
)
# Always render in light mode (white background), even when the visitor's system is dark.
FORCE_LIGHT = """
() => {
const url = new URL(window.location.href);
if (url.searchParams.get('__theme') !== 'light') {
url.searchParams.set('__theme', 'light');
window.location.replace(url.href);
}
}
"""
with gr.Blocks(title="Audio Insight", delete_cache=(3600, 7200)) as demo:
gr.Markdown(INTRO)
with gr.Row(equal_height=False):
with gr.Column(scale=1):
files = gr.File(label="Audio files", file_count="multiple",
file_types=["audio", "video"] + config.AUDIO_EXTENSIONS, height=220)
with gr.Column(scale=1):
context = gr.Textbox(
label="What are these recordings? (optional)", lines=3, max_lines=8,
placeholder="e.g. Research interviews with nurses about AI documentation tools. "
"Names and technical terms help the transcript.")
questions = gr.Textbox(
label="Questions to answer across the recordings (optional, one per line)", lines=3, max_lines=10,
placeholder="What pain points do participants describe?\nHow do they feel about AI?")
with gr.Accordion("Options", open=False):
with gr.Row():
use_ai = gr.Checkbox(value=True, label="Use Gemini (transcript, themes, sentiment, cross-file patterns)")
deep = gr.Checkbox(value=False, label="Deeper reasoning (Gemini Pro writes the analysis; slower)")
api_key = gr.Textbox(label="Gemini API key (optional)", type="password",
placeholder="Leave empty to use the Space's key. Your key is used for this run only and never stored.")
run_btn = gr.Button("Analyse", variant="primary", size="lg")
status = gr.Markdown()
report_view = gr.HTML()
downloads = gr.File(label="Downloads", file_count="multiple", interactive=False)
run_btn.click(analyse, [files, context, questions, use_ai, deep, api_key], [status, report_view, downloads],
concurrency_limit=2, show_progress="full")
if __name__ == "__main__":
demo.queue(default_concurrency_limit=2, max_size=20)
demo.launch(css=CSS, theme=THEME, js=FORCE_LIGHT, max_file_size=f"{config.MAX_UPLOAD_MB}mb",
allowed_paths=[OUT_ROOT])