"""Transcribe MP3 attachments (HF Inference, local whisper, or sidecar txt).""" from __future__ import annotations import os import re from pathlib import Path from typing import Optional TRANSCRIPT_DIR = Path(__file__).resolve().parent.parent / "files" / "transcripts" def _sidecar_transcript(path: Path) -> Optional[str]: sidecars = [ TRANSCRIPT_DIR / f"{path.stem}.txt", path.with_suffix(".txt"), ] for sc in sidecars: if sc.is_file(): return sc.read_text(encoding="utf-8").strip() return None def transcribe(path: Path) -> Optional[str]: text = _sidecar_transcript(path) if text: return text token = os.environ.get("HF_TOKEN") or os.environ.get("HUGGINGFACEHUB_API_TOKEN") if token: try: from huggingface_hub import InferenceClient client = InferenceClient(token=token) with open(path, "rb") as f: out = client.automatic_speech_recognition(f, model="openai/whisper-large-v3") if isinstance(out, dict): return (out.get("text") or "").strip() return str(out).strip() except Exception: pass try: import whisper model = whisper.load_model("tiny") result = model.transcribe(str(path)) return (result.get("text") or "").strip() except Exception: return None def strawberry_pie_ingredients(transcript: str) -> str: # Pull filling ingredients; alphabetize; no measurements # Known pattern from recipe audio candidates = [] patterns = [ r"ripe strawberries", r"granulated sugar", r"freshly squeezed lemon juice", r"cornstarch", r"pure vanilla extract", r"\bsalt\b", r"butter", ] lower = transcript.lower() for p in patterns: if re.search(p, lower): # normalize name from pattern name = p.replace(r"\b", "").replace("\\", "") candidates.append(name) # Prefer explicit ordered extraction from combine clause m = re.search( r"combine ([^.]+?)(?:\.|Cook)", transcript, flags=re.I, ) items = [] if m: chunk = m.group(1) # split on commas and and parts = re.split(r",| and ", chunk) items = [p.strip().lower() for p in parts if p.strip()] # vanilla separately if "vanilla" in lower: for phrase in ("pure vanilla extract", "vanilla extract", "vanilla"): if phrase in lower and phrase not in items: items.append("pure vanilla extract" if "pure vanilla" in lower else phrase) break # Deduplicate preserving canonical names canon = [] for it in items: it = it.strip(" .") if it and it not in canon: canon.append(it) if not canon: canon = sorted(set(candidates)) return ", ".join(sorted(canon)) def calculus_pages(transcript: str) -> str: pages = set() # Matches: "page 245", "pages 132, 133, and 134", "On page 132, 133 and 134" for m in re.finditer( r"pages?\s+((?:\d+(?:\s*,\s*|\s+and\s+|\s+)*)+\d+|\d+)", transcript, flags=re.I, ): pages.update(int(x) for x in re.findall(r"\d+", m.group(1))) # Fallback: any 3-digit number near "page" if len(pages) < 3: for m in re.finditer(r"page[^.]{0,40}?(\d{3})", transcript, flags=re.I): pages.add(int(m.group(1))) pages.update(int(x) for x in re.findall(r"\b(1[3-9]\d|2\d{2})\b", transcript)) return ", ".join(str(p) for p in sorted(pages))