Spaces:
Running on Zero
Running on Zero
Download scripts/omniroute_workflows_smoke.py from abalanescu/flow2: direct link, hf CLI and curl.
- Browser
- Download file 4.92 kB
-
https://huggingface.co/spaces/abalanescu/flow2/resolve/main/scripts/omniroute_workflows_smoke.py
- Command line
-
hf download hf://spaces/abalanescu/flow2/scripts/omniroute_workflows_smoke.py
-
curl -L -o omniroute_workflows_smoke.py https://huggingface.co/spaces/abalanescu/flow2/resolve/main/scripts/omniroute_workflows_smoke.py
4.92 kB
| #!/usr/bin/env python3 | |
| """omniroute_workflows_smoke.py — Live smoke test for OmniRoute workflows. | |
| Exercises the 4 Milestone 5 line 69 workflows against the live gateway: | |
| 1. Semantic search (infinity/BAAI/bge-m3) | |
| 2. Meeting transcription (whisperfw/Systran/faster-whisper-small) | |
| 3. Structured extraction (1zero combo) | |
| 4. Document summarization (1zero combo) | |
| Requires OMNIROUTE_API_KEY environment variable (set it before running; | |
| there is no hardcoded default). Safe to run repeatedly; creates a scratch | |
| temporary WAV file and cleans it up. | |
| """ | |
| from __future__ import annotations | |
| import os | |
| import sys | |
| import tempfile | |
| import time | |
| import wave | |
| # Ensure moldovan-qwen is on path | |
| ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), "..")) | |
| sys.path.insert(0, os.path.join(ROOT, "moldovan-qwen")) | |
| from workflows.providers import OmniRouteClient, OmniRouteError # noqa: E402 | |
| from workflows.summarizer import summarize_document # noqa: E402 | |
| from workflows.transcriber import transcribe_meeting # noqa: E402 | |
| from workflows.semantic_search import SemanticSearchIndex # noqa: E402 | |
| from workflows.extractor import extract_structured_data # noqa: E402 | |
| API_KEY = os.environ.get("OMNIROUTE_API_KEY", "") | |
| BASE_URL = os.environ.get( | |
| "OMNIROUTE_BASE_URL", "http://homelab:20128" | |
| ) | |
| def _make_dummy_wav(path: str) -> None: | |
| with wave.open(path, "w") as w: | |
| w.setnchannels(1) | |
| w.setsampwidth(2) | |
| w.setframerate(16000) | |
| # 0.5s of near-silence | |
| w.writeframes(b"\x00\x00" * 8000) | |
| def main() -> int: | |
| client = OmniRouteClient(base_url=BASE_URL, api_key=API_KEY, timeout=240.0) | |
| print(f"[smoke] Target gateway: {BASE_URL}") | |
| # 1. Semantic Search | |
| print("\n[1/4] Testing semantic search (infinity/BAAI/bge-m3)...") | |
| t0 = time.time() | |
| try: | |
| idx = SemanticSearchIndex(client) | |
| docs = [ | |
| "Placinta cu branza si marar coapta la cuptor.", | |
| "Dezvoltarea retelelor neuronale pentru procesarea limbajului natural.", | |
| "Vin moldovenesc Feteasca Neagra maturat in butoaie de stejar.", | |
| ] | |
| n = idx.add_documents(docs) | |
| print(f" Indexed {n} docs in {time.time() - t0:.1f}s") | |
| results = idx.search("mancare traditionala moldoveneasca", top_k=1) | |
| assert results, "no search results" | |
| print(f" Query matched: {results[0]['document'][:50]}... (score: {results[0]['score']:.3f})") | |
| print(" PASS") | |
| except Exception as exc: | |
| print(f" FAIL: {exc}") | |
| return 1 | |
| # 2. Meeting Transcription | |
| print("\n[2/4] Testing meeting transcription (whisperfw)...") | |
| t0 = time.time() | |
| with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as tmp: | |
| tmp_wav = tmp.name | |
| try: | |
| _make_dummy_wav(tmp_wav) | |
| res = transcribe_meeting([tmp_wav], client) | |
| print(f" Transcribed in {time.time() - t0:.1f}s: '{res['transcript']}'") | |
| print(" PASS") | |
| except Exception as exc: | |
| print(f" FAIL: {exc}") | |
| return 2 | |
| finally: | |
| try: | |
| os.remove(tmp_wav) | |
| except OSError: | |
| pass | |
| # 3. Structured Extraction | |
| print("\n[3/4] Testing structured extraction (1zero)...") | |
| t0 = time.time() | |
| text = ( | |
| "Clientul Ion Creanga, nascut pe 1 martie 1837 in Humulesti, " | |
| "a comandat 3 saci de faina si 2 butoaie de vin." | |
| ) | |
| schema = { | |
| "nume": "string", | |
| "locul_nasterii": "string", | |
| "produse": "lista de stringuri sau obiecte", | |
| } | |
| try: | |
| extracted = extract_structured_data(text, schema, client) | |
| print(f" Extracted in {time.time() - t0:.1f}s: {extracted}") | |
| assert isinstance(extracted, dict) and "nume" in extracted | |
| print(" PASS") | |
| except Exception as exc: | |
| print(f" FAIL: {exc}") | |
| return 3 | |
| # 4. Document Summarization | |
| print("\n[4/4] Testing document summarization (1zero)...") | |
| t0 = time.time() | |
| doc = ( | |
| "Republica Moldova este un stat situat in Europa de Est, " | |
| "invecinat cu Romania la vest si Ucraina la nord, est si sud. " | |
| "Capitala tarii este Chisinau. Economia se bazeaza pe agricultura, " | |
| "industria vinicola si sectorul serviciilor IT in rapida expansiune. " | |
| "Traditiile populare, muzica si ospitalitatea moldoveneasca sunt " | |
| "recunoscute la nivel international." | |
| ) | |
| try: | |
| summary_res = summarize_document(doc, client) | |
| print(f" Summarized in {time.time() - t0:.1f}s: {summary_res['summary']}") | |
| assert summary_res["summary"], "empty summary" | |
| print(" PASS") | |
| except Exception as exc: | |
| print(f" FAIL: {exc}") | |
| return 4 | |
| print("\n" + "=" * 50) | |
| print(" ALL 4 OMNIROUTE WORKFLOWS PASSED LIVE SMOKE CHECK") | |
| print("=" * 50) | |
| return 0 | |
| if __name__ == "__main__": | |
| sys.exit(main()) |