"""End to end: full small GGUF + real cast + CPU inference. Needs VOICE_TEST_LIVE=1 and llama-cli. ~700 MB model download (cached), then local. """ import json import subprocess import sys import tempfile import unittest from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parent)) from _helpers import BASE_FILE, BASE_REPO, LLAMA_CLI # noqa: E402 from _helpers import cache_dir, fetch_file, needs_llama, ns, voice # noqa: E402 def shard_url(repo, filename): return f"https://huggingface.co/{repo}/resolve/main/{filename}" @needs_llama class TestEndToEnd(unittest.TestCase): @classmethod def setUpClass(cls): cls.work = Path(tempfile.mkdtemp(prefix="e2e-")) cls._old = voice.VOICES_DIR voice.VOICES_DIR = cls.work / "voices" # shared live voices from test_voices? No: self-contained — refetch head here # via the cached-voice path would couple suites; instead reuse cache dir voices # fetched by a direct cmd_get (second suite to do so stays independent). voice.cmd_get(ns(source="xcx0902/Qwen3-1.7B-catgirl", name="e2e-catgirl")) cls.model = fetch_file(shard_url(BASE_REPO, BASE_FILE), cache_dir() / BASE_FILE, label=BASE_FILE, n_parts=8) cls.voiced = cls.work / "voiced.gguf" voice.cmd_cast(ns(target=str(cls.model), voice="e2e-catgirl", out=str(cls.voiced))) @classmethod def tearDownClass(cls): voice.VOICES_DIR = cls._old def test_cast_replaces_head_only(self): from gguf import GGUFReader head = "token_embd.weight" # Qwen3 ties the head: no output.weight in GGUF before = {t.name: t.tensor_type.name for t in GGUFReader(str(self.model)).tensors} after_rd = GGUFReader(str(self.voiced)) after = {t.name: t.tensor_type.name for t in after_rd.tensors} self.assertEqual(set(before), set(after)) self.assertIn(head, after) self.assertEqual(after[head], "Q8_0") for n, q in before.items(): if n != head: self.assertEqual(after[n], q, n) # values actually changed on the head b = [t for t in GGUFReader(str(self.model)).tensors if t.name == head][0] a = [t for t in after_rd.tensors if t.name == head][0] self.assertFalse((bytes(a.data) == bytes(b.data))) def test_inference_runs(self): from _helpers import run_guarded prompt = "The tavern door creaked open and" r = run_guarded( [LLAMA_CLI, "-m", str(self.voiced), "-p", prompt, "-n", "24", "--seed", "42", "-t", "8", "-c", "512", "--no-display-prompt", "--single-turn", "-e"], timeout=600) self.assertEqual(r.returncode, 0, (r.stdout + r.stderr)[-2000:]) gen = r.stdout.strip() self.assertGreater(len(gen), 0) print(f"\n[llama] {gen[:200]}") if __name__ == "__main__": unittest.main()