audio-to-notes / app.py
BonusLockSMith's picture
UI: GritAI engineered/industrial design system (dark, grit-orange, Archivo+IBM Plex)
38097c0 verified
Raw History Blame Contribute Delete
12.1 kB
#!/usr/bin/env python
"""
GritAI β€” Audio β†’ Notes (Project #7 of "30 AI Projects in 15 Days")
Upload a recording or record from the mic -> a clean TRANSCRIPT with per-minute [mm:ss] timestamps
(quotable by the minute) -> a short SUMMARY + a list of ACTION ITEMS (task / owner / due).
FREE tier: faster-whisper ("small") on HF ZeroGPU for transcription + one Claude (haiku) call with
forced tool-use for schema-valid {summary, action_items}. PRO tier ("GritAI Studio"): long meetings,
SPEAKER attribution (who said what), a searchable/quotable archive, and auto-generated briefing docs --
the composition of several 30-in-15 rungs, assembled later. Shown here as a waitlist.
Two-stage pipeline: transcription runs inside @spaces.GPU (GPU held only for that); the summarization
Claude call runs after, on CPU, so the GPU is released before the network call.
"""
import os
import gradio as gr
ON_SPACE = bool(os.environ.get("SPACE_ID")) # True when running on a Hugging Face Space
WHISPER_SIZE = "base" # int8 on CPU β€” fast + reliable for short clips, no CUDA libs needed
CLAUDE_MODEL = os.environ.get("ANTHROPIC_MODEL", "claude-haiku-4-5")
MAX_TRANSCRIPT_CHARS = 24000 # keep the summarize call comfortably in-context
_MODEL = None
def _get_model():
"""Construct the faster-whisper model on CPU (int8), cached. Runs anywhere β€” no GPU / CUDA libs."""
global _MODEL
if _MODEL is None:
from faster_whisper import WhisperModel
_MODEL = WhisperModel(WHISPER_SIZE, device="cpu", compute_type="int8")
return _MODEL
# Warm the model at startup on the Space so the first request isn't slow.
if ON_SPACE:
try:
_get_model()
print("[startup] faster-whisper (base, cpu/int8) loaded", flush=True)
except Exception as _e:
print(f"[startup] whisper load failed (will retry on first call): {_e}", flush=True)
def _mmss(seconds):
seconds = int(seconds or 0)
return f"{seconds // 60:02d}:{seconds % 60:02d}"
def _transcribe(audio_path):
try:
model = _get_model()
except ImportError:
raise gr.Error(
"πŸŽ™οΈ Local UI preview β€” transcription runs on the Space (faster-whisper isn't installed here). "
"The controls and Studio waitlist are live."
)
segments, info = model.transcribe(audio_path, vad_filter=True, beam_size=5)
rows = [(seg.start, seg.end, seg.text.strip()) for seg in segments]
return rows, getattr(info, "language", "?")
NOTES_TOOL = {
"name": "emit_notes",
"description": "Return a concise summary and the action items from a meeting/voice-memo transcript.",
"input_schema": {
"type": "object",
"properties": {
"summary": {"type": "string", "description": "3-5 sentence plain-English summary."},
"action_items": {
"type": "array",
"items": {
"type": "object",
"properties": {
"task": {"type": "string"},
"owner": {"type": "string", "description": "Person responsible, or 'unassigned'."},
"due": {"type": "string", "description": "Due date/'when', or 'not stated'."},
},
"required": ["task"],
},
},
},
"required": ["summary", "action_items"],
},
}
def _summarize(transcript):
key = os.environ.get("ANTHROPIC_API_KEY")
if not key:
return "_(Set ANTHROPIC_API_KEY on the Space to get the summary + action items.)_", ""
try:
import anthropic
client = anthropic.Anthropic(api_key=key)
msg = client.messages.create(
model=CLAUDE_MODEL,
max_tokens=1024,
tools=[NOTES_TOOL],
tool_choice={"type": "tool", "name": "emit_notes"},
messages=[{"role": "user", "content":
"Summarize this transcript and extract the action items.\n\n" + transcript[:MAX_TRANSCRIPT_CHARS]}],
)
data = next((b.input for b in msg.content if getattr(b, "type", "") == "tool_use"), {})
except Exception as e:
return f"_(Summary unavailable: {type(e).__name__})_", ""
summary = data.get("summary", "").strip()
items = data.get("action_items", []) or []
if items:
lines = []
for it in items:
task = (it.get("task") or "").strip()
owner = (it.get("owner") or "unassigned").strip()
due = (it.get("due") or "not stated").strip()
lines.append(f"- **{task}** β€” _{owner}_ Β· {due}")
actions_md = "\n".join(lines)
else:
actions_md = "_No action items found._"
return summary or "_(no summary)_", actions_md
def process(audio_path, progress=gr.Progress(track_tqdm=True)):
if not audio_path:
raise gr.Error("Upload a file or record from the mic first.")
rows, lang = _transcribe(audio_path)
if not rows:
raise gr.Error("No speech detected in that audio.")
transcript = "\n".join(f"[{_mmss(s)}] {text}" for s, e, text in rows if text)
summary, actions_md = _summarize(transcript)
return transcript, summary, actions_md
# ----------------------------------------------------------------------------- UI
THEME = gr.themes.Base(
primary_hue=gr.themes.colors.orange, neutral_hue=gr.themes.colors.slate,
font=[gr.themes.GoogleFont("IBM Plex Sans"), "system-ui", "sans-serif"],
font_mono=[gr.themes.GoogleFont("IBM Plex Mono"), "monospace"],
).set(
body_background_fill="#0a0b0e", body_text_color="#eef0f4", body_text_color_subdued="#9aa1b0",
background_fill_primary="#111318", background_fill_secondary="#0e1015",
block_background_fill="#111318", block_border_color="#20242e",
block_label_background_fill="#0e1015", block_label_text_color="#9aa1b0", block_title_text_color="#eef0f4",
border_color_primary="#20242e", border_color_accent="#c9451f",
input_background_fill="#0e1015", input_border_color="#2b303c", input_border_color_focus="#c9451f",
button_primary_background_fill="#ff5a2a", button_primary_background_fill_hover="#ff7d54",
button_primary_text_color="#140a06", button_primary_border_color="#c9451f",
button_secondary_background_fill="#15181f", button_secondary_text_color="#eef0f4",
button_secondary_border_color="#2b303c",
color_accent="#ff5a2a", color_accent_soft="#2b1810", slider_color="#ff5a2a",
checkbox_background_color_selected="#ff5a2a", checkbox_border_color_focus="#c9451f",
)
CSS = """
@import url('https://fonts.googleapis.com/css2?family=Archivo:wght@700;800;900&family=IBM+Plex+Mono:wght@400;500;600&family=IBM+Plex+Sans:wght@400;500;600&display=swap');
.gradio-container{background:#0a0b0e !important}
.gradio-container::before{content:"";position:fixed;inset:0;z-index:0;pointer-events:none;
background:repeating-linear-gradient(90deg,transparent 0 63px,rgba(255,255,255,.018) 63px 64px),
repeating-linear-gradient(0deg,transparent 0 63px,rgba(255,255,255,.018) 63px 64px),
radial-gradient(120% 70% at 50% -10%,#141821 0%,#0a0b0e 55%)}
.main,.contain{position:relative;z-index:1}
#ghead{display:flex;align-items:center;gap:14px;padding:6px 2px 4px}
#ghead .co{font-family:'Archivo',sans-serif;font-weight:900;font-size:19px;letter-spacing:.5px;color:#eef0f4;line-height:1}
#ghead .co b{color:#ff5a2a}
#ghead .sub{font-family:'IBM Plex Mono',monospace;font-size:10.5px;letter-spacing:.28em;text-transform:uppercase;color:#5f6675;margin-top:5px}
#ghead .tags{margin-left:auto;display:flex;gap:8px}
#ghead .tag{font-family:'IBM Plex Mono',monospace;font-size:10px;letter-spacing:.14em;text-transform:uppercase;color:#9aa1b0;border:1px solid #2b303c;border-radius:2px;padding:5px 9px}
#ghead .tag.hot{color:#ff5a2a;border-color:#c9451f}
#ghero h1{font-family:'Archivo',sans-serif;font-weight:800;font-size:clamp(26px,4.2vw,40px);line-height:1.05;letter-spacing:-.02em;margin:14px 0 4px;color:#eef0f4}
#ghero h1 .g{color:#ff5a2a}
#ghero p{color:#9aa1b0;margin:0;font-size:14.5px}
footer{visibility:hidden}
button.selected{color:#ff5a2a !important}
.gritai-foot{text-align:center;color:#5f6675;font-size:.8rem;margin-top:16px;font-family:'IBM Plex Mono',monospace;letter-spacing:.05em}
.gritai-foot a{color:#9aa1b0;text-decoration:none}
.pro-card{border:1px solid #2b303c;border-left:2px solid #ff5a2a;border-radius:4px;padding:16px 18px;background:#111318}
.pro-card h3{color:#eef0f4}
"""
with gr.Blocks(theme=THEME, css=CSS, title="GritAI Β· Audio β†’ Notes") as demo:
gr.HTML('''<div id="ghead">
<div style="width:34px;height:34px;flex:none"><svg width="34" height="34" viewBox="0 0 38 38" fill="none"><rect x="1" y="1" width="36" height="36" rx="2" stroke="#2b303c"/><path d="M27 12.5A9 9 0 1 0 28 22H19" stroke="#ff5a2a" stroke-width="2.4" stroke-linecap="square"/><rect x="10.5" y="10.5" width="3" height="3" fill="#ff5a2a"/></svg></div>
<div><div class="co">GRIT<b>AI</b></div><div class="sub">Audio &rarr; Notes</div></div>
<div class="tags"><span class="tag hot">&#9656; Live</span><span class="tag">SDVOSB</span><span class="tag">Lawton OK</span></div>
</div>
<div id="ghero"><h1>Audio in. <span class="g">Notes</span> out.</h1>
<p>Timestamped transcript + summary + action items &middot; quotable by the minute &middot; Project #7 of 30 in 15 days.</p></div>''')
with gr.Tab("Transcribe & summarize (free)"):
with gr.Row():
with gr.Column(scale=2):
audio = gr.Audio(sources=["upload", "microphone"], type="filepath", label="Audio")
go = gr.Button("Transcribe β–Ά", variant="primary")
gr.Markdown("<small>Best on short clips (a few minutes). Long meetings + speaker labels "
"are the Pro tier.</small>")
with gr.Column(scale=3):
summary = gr.Markdown(label="Summary")
actions = gr.Markdown(label="Action items")
transcript = gr.Textbox(label="Transcript ( [mm:ss] per line )", lines=16, show_copy_button=True)
go.click(process, inputs=[audio], outputs=[transcript, summary, actions], api_name="transcribe")
with gr.Tab("GritAI Studio (Pro)"):
gr.Markdown(
"<div class='pro-card'>\n\n"
"### πŸš€ Beyond the free demo\n"
"The free tier transcribes short clips and pulls a summary + action items. **GritAI Studio** "
"turns audio into an actual knowledge base:\n\n"
"- **Long meetings** β€” hours, not minutes.\n"
"- **Speaker attribution** β€” who said what, quotable by **minute _and_ speaker**.\n"
"- **Searchable & quotable archive** β€” find any moment across every recording, with citations.\n"
"- **Auto-briefings** β€” one-click briefing docs generated from the transcript.\n\n"
"*Built on our own stack β€” the composition of several GritAI tools into one briefing room.*\n\n"
"</div>"
)
with gr.Row():
email = gr.Textbox(label="Email", placeholder="you@team.com", scale=3)
use_case = gr.Textbox(label="What would you use it for?", placeholder="e.g. weekly stand-ups", scale=4)
join = gr.Button("Join the Studio waitlist", variant="primary")
waitlist_msg = gr.Markdown(visible=False)
def _join(email_v, use_case_v):
if not email_v or "@" not in email_v:
return gr.update(value="⚠️ Please enter a valid email.", visible=True)
return gr.update(value="βœ… You're on the list β€” we'll reach out when GritAI Studio opens. Thanks!",
visible=True)
join.click(_join, inputs=[email, use_case], outputs=waitlist_msg)
gr.HTML("<div class='gritai-foot'>faster-whisper + Claude &middot; "
"<a href='https://gritai.solutions'>GRITAI SOLUTIONS</a> &middot; 30-in-15 &#8470;7</div>")
if __name__ == "__main__":
demo.queue(max_size=20).launch(server_name="0.0.0.0", server_port=int(os.environ.get("PORT", 7860)))