File size: 4,915 Bytes
7036d8f
 
 
 
 
 
 
 
 
91f1e9a
 
7036d8f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
91f1e9a
7036d8f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
#!/usr/bin/env python3
"""omniroute_workflows_smoke.py — Live smoke test for OmniRoute workflows.

Exercises the 4 Milestone 5 line 69 workflows against the live gateway:
  1. Semantic search (infinity/BAAI/bge-m3)
  2. Meeting transcription (whisperfw/Systran/faster-whisper-small)
  3. Structured extraction (1zero combo)
  4. Document summarization (1zero combo)

Requires OMNIROUTE_API_KEY environment variable (set it before running;
there is no hardcoded default). Safe to run repeatedly; creates a scratch
temporary WAV file and cleans it up.
"""

from __future__ import annotations

import os
import sys
import tempfile
import time
import wave

# Ensure moldovan-qwen is on path
ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), ".."))
sys.path.insert(0, os.path.join(ROOT, "moldovan-qwen"))

from workflows.providers import OmniRouteClient, OmniRouteError  # noqa: E402
from workflows.summarizer import summarize_document  # noqa: E402
from workflows.transcriber import transcribe_meeting  # noqa: E402
from workflows.semantic_search import SemanticSearchIndex  # noqa: E402
from workflows.extractor import extract_structured_data  # noqa: E402

API_KEY = os.environ.get("OMNIROUTE_API_KEY", "")
BASE_URL = os.environ.get(
    "OMNIROUTE_BASE_URL", "http://homelab:20128"
)


def _make_dummy_wav(path: str) -> None:
    with wave.open(path, "w") as w:
        w.setnchannels(1)
        w.setsampwidth(2)
        w.setframerate(16000)
        # 0.5s of near-silence
        w.writeframes(b"\x00\x00" * 8000)


def main() -> int:
    client = OmniRouteClient(base_url=BASE_URL, api_key=API_KEY, timeout=240.0)
    print(f"[smoke] Target gateway: {BASE_URL}")

    # 1. Semantic Search
    print("\n[1/4] Testing semantic search (infinity/BAAI/bge-m3)...")
    t0 = time.time()
    try:
        idx = SemanticSearchIndex(client)
        docs = [
            "Placinta cu branza si marar coapta la cuptor.",
            "Dezvoltarea retelelor neuronale pentru procesarea limbajului natural.",
            "Vin moldovenesc Feteasca Neagra maturat in butoaie de stejar.",
        ]
        n = idx.add_documents(docs)
        print(f"      Indexed {n} docs in {time.time() - t0:.1f}s")
        results = idx.search("mancare traditionala moldoveneasca", top_k=1)
        assert results, "no search results"
        print(f"      Query matched: {results[0]['document'][:50]}... (score: {results[0]['score']:.3f})")
        print("      PASS")
    except Exception as exc:
        print(f"      FAIL: {exc}")
        return 1

    # 2. Meeting Transcription
    print("\n[2/4] Testing meeting transcription (whisperfw)...")
    t0 = time.time()
    with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as tmp:
        tmp_wav = tmp.name
    try:
        _make_dummy_wav(tmp_wav)
        res = transcribe_meeting([tmp_wav], client)
        print(f"      Transcribed in {time.time() - t0:.1f}s: '{res['transcript']}'")
        print("      PASS")
    except Exception as exc:
        print(f"      FAIL: {exc}")
        return 2
    finally:
        try:
            os.remove(tmp_wav)
        except OSError:
            pass

    # 3. Structured Extraction
    print("\n[3/4] Testing structured extraction (1zero)...")
    t0 = time.time()
    text = (
        "Clientul Ion Creanga, nascut pe 1 martie 1837 in Humulesti, "
        "a comandat 3 saci de faina si 2 butoaie de vin."
    )
    schema = {
        "nume": "string",
        "locul_nasterii": "string",
        "produse": "lista de stringuri sau obiecte",
    }
    try:
        extracted = extract_structured_data(text, schema, client)
        print(f"      Extracted in {time.time() - t0:.1f}s: {extracted}")
        assert isinstance(extracted, dict) and "nume" in extracted
        print("      PASS")
    except Exception as exc:
        print(f"      FAIL: {exc}")
        return 3

    # 4. Document Summarization
    print("\n[4/4] Testing document summarization (1zero)...")
    t0 = time.time()
    doc = (
        "Republica Moldova este un stat situat in Europa de Est, "
        "invecinat cu Romania la vest si Ucraina la nord, est si sud. "
        "Capitala tarii este Chisinau. Economia se bazeaza pe agricultura, "
        "industria vinicola si sectorul serviciilor IT in rapida expansiune. "
        "Traditiile populare, muzica si ospitalitatea moldoveneasca sunt "
        "recunoscute la nivel international."
    )
    try:
        summary_res = summarize_document(doc, client)
        print(f"      Summarized in {time.time() - t0:.1f}s: {summary_res['summary']}")
        assert summary_res["summary"], "empty summary"
        print("      PASS")
    except Exception as exc:
        print(f"      FAIL: {exc}")
        return 4

    print("\n" + "=" * 50)
    print(" ALL 4 OMNIROUTE WORKFLOWS PASSED LIVE SMOKE CHECK")
    print("=" * 50)
    return 0


if __name__ == "__main__":
    sys.exit(main())