Download app.py from jer3mi/loudkit: direct link, hf CLI and curl.
- Browser
- Download file 31.4 kB
-
https://huggingface.co/spaces/jer3mi/loudkit/resolve/main/app.py
- Command line
-
hf download hf://spaces/jer3mi/loudkit/app.py
-
curl -L -o app.py https://huggingface.co/spaces/jer3mi/loudkit/resolve/main/app.py
31.4 kB
| """The loudkit demo Space: twenty-eight voices, two models, one engine. | |
| ZeroGPU bills GPU time to the *visitor*, not to the owner: an anonymous visitor | |
| gets about two minutes a day, a signed-in free account about five. A demo whose | |
| first click spends that budget is one most people bounce off before they have | |
| heard anything at all. So the Voices tab is fifty-six pre-rendered files served | |
| straight out of this repo, twenty-eight voices on each model, and the GPU is | |
| spent only on what a visitor types or records. | |
| Both models are here because the choice between them is the first real decision | |
| an adopter makes, and it is not one a table of numbers settles. Every voice | |
| reads the same passage on both, at the same seed, so switching the model is the | |
| only thing that changes. loudr-1 is the natural one; loudr-1-turbo renders | |
| about twice as fast from a two-token decode. | |
| Both are loaded at module level on `cuda`, which is what ZeroGPU asks for: CUDA | |
| transfers are optimised for start-up placement, and lazy-loading inside a | |
| `@spaces.GPU` function is explicitly discouraged. Each decorated call then runs | |
| in a freshly forked, short-lived process, which is also why there is no | |
| `torch.compile` and no CUDA graph capture here: both pay their cost once per | |
| process and would never amortise. | |
| Cloning is exposed, and the consent is built into the shape of the tab rather | |
| than written beside it: the microphone is the default path, an upload is | |
| secondary and gated on an explicit confirmation, and neither recording outlives | |
| the request that carried it. | |
| Every load is pinned to a Hub revision. A demo that says what a build does must | |
| be able to say which build, and `main` is not an answer that stays true. | |
| """ | |
| from __future__ import annotations | |
| import contextlib | |
| import dataclasses | |
| import hashlib | |
| import json | |
| import os | |
| import tempfile | |
| from pathlib import Path | |
| # Before torch, and before anything that imports torch. The module installs the | |
| # CUDA emulation that lets a module-level `.to("cuda")` succeed on a machine | |
| # that has no GPU attached yet. | |
| import spaces | |
| import gradio as gr | |
| import numpy as np | |
| import loudkit as lk | |
| from loudkit.backends.torch_backend import build_torch_enroller | |
| from loudkit.hub import resolve_enrollment_checkpoint, resolve_voice_encoder | |
| DEVICE = "cuda" | |
| HERE = Path(__file__).parent | |
| # The two published bundles, pinned. The tag moves when a release is cut; the | |
| # commit does not, and a demo that claims a fingerprint should be reproducible | |
| # from the same bytes a year from now. | |
| MODELS = { | |
| "loudr-1": { | |
| "repo": "loudreader/loudr-1", | |
| "revision": "7516673ee74ab2228ffc4e99223d296d8904d7ff", | |
| "label": "loudr-1 · the natural one", | |
| "blurb": "Single-token decode. The reference model.", | |
| }, | |
| "turbo": { | |
| "repo": "loudreader/loudr-1-turbo", | |
| "revision": "366f140a5f7ebe215e5b43cdd73384746c83bae6", | |
| "label": "loudr-1-turbo · about twice as fast", | |
| "blurb": "Two-token decode, one flow step. Same voices, same profiles.", | |
| }, | |
| } | |
| DEFAULT_MODEL = "loudr-1" | |
| MODEL_CHOICES = [(cfg["label"], key) for key, cfg in MODELS.items()] | |
| # The CPU scaffold capped text at 300 characters because CPU synthesis ran at | |
| # roughly a tenth of real time. On a GPU the cap is about the visitor's daily | |
| # quota instead, which is a far looser bound: 1000 characters is ~70 s of speech. | |
| MAX_CHARS = 1_000 | |
| MAX_CLONE_CHARS = 400 | |
| MAX_PROBE_CHARS = 200 | |
| # The enroller refuses anything over 30 s and wants 5 to 10. The prompt is built | |
| # from the first 10 s; the speaker embedding reads whatever else is there, so a | |
| # little past the prompt window is useful and 20 s stays clear of the refusal. | |
| ENROLL_SECONDS = 20.0 | |
| DOCS = "https://github.com/loudreader/loudkit" | |
| IDENTITY_CONTRACT = f"{DOCS}/blob/main/docs/reference/IDENTITY-CONTRACT.md" | |
| RESPONSIBLE_USE = f"{DOCS}/blob/main/RESPONSIBLE_USE.md" | |
| ROSTER = json.loads((HERE / "voices.json").read_text(encoding="utf-8")) | |
| BY_NAME = {entry["name"]: entry for entry in ROSTER} | |
| ORDERED = sorted(ROSTER, key=lambda e: (e["language"], e["name"])) | |
| # English leads. A visitor who does not read the other nine should not have to | |
| # hunt for the one they do, and alphabetical order put Danish first. | |
| FIRST_LANGUAGE = "en" | |
| _LANG_OF = {e["language_id"]: e["language"] for e in ROSTER} | |
| LANGUAGE_FILTER = [(_LANG_OF[FIRST_LANGUAGE], FIRST_LANGUAGE)] + [ | |
| (name, code) | |
| for code, name in sorted(_LANG_OF.items(), key=lambda kv: kv[1]) | |
| if code != FIRST_LANGUAGE | |
| ] | |
| def voices_in(language: str) -> list[tuple[str, str]]: | |
| # Z to A within a language; the first entry is the voice that plays on landing. | |
| return [ | |
| (f"{e['name']} ({e['gender']})", e["name"]) | |
| for e in sorted(ORDERED, key=lambda e: e["name"], reverse=True) | |
| if e["language_id"] == language | |
| ] | |
| FIRST_VOICES = voices_in(FIRST_LANGUAGE) | |
| FIRST_VOICE = FIRST_VOICES[0][1] | |
| # Reference clips a visitor can clone without recording anything. Every one was | |
| # donated for building TTS voices, so these need no consent box; the roster | |
| # carries the licence and the donor's own words beside each. | |
| CLONE_EXAMPLES = ["joe", "kathleen", "ines", "gosia"] | |
| EXAMPLE_CHOICES = [("Nothing selected", "")] + [ | |
| (f"{n} · {BY_NAME[n]['language']}", n) for n in CLONE_EXAMPLES | |
| ] | |
| # -------------------------------------------------------------------------- | |
| # Module-level model placement, per the ZeroGPU contract. | |
| # -------------------------------------------------------------------------- | |
| ENGINES = { | |
| key: lk.load(cfg["repo"], revision=cfg["revision"], device=DEVICE) | |
| for key, cfg in MODELS.items() | |
| } | |
| # Voice profiles are numpy, not torch, so they are device-agnostic and cost a | |
| # few hundred kilobytes each. They are also identical in both bundles, which is | |
| # the point of a portable profile: one clone, either model. | |
| BASE = MODELS[DEFAULT_MODEL] | |
| PROFILES = { | |
| entry["name"]: lk.voice(entry["name"], repo=BASE["repo"], revision=BASE["revision"]) | |
| for entry in ROSTER | |
| } | |
| # Enrollment reads the other half of the release: the speech tokenizer and the | |
| # speaker encoder, which synthesis never touches, plus the utterance voice | |
| # encoder that sits beside both. One enroller serves both models, because the | |
| # profile it produces is the same file either way. `lk.enroll()` builds this per | |
| # call by design; a Space would pay the load on every clone. | |
| enroller = build_torch_enroller( | |
| str(resolve_enrollment_checkpoint(BASE["repo"], revision=BASE["revision"])), | |
| device=DEVICE, | |
| voice_encoder_weights=str(resolve_voice_encoder(BASE["repo"], revision=BASE["revision"])), | |
| ) | |
| FINGERPRINTS = {key: engine.algorithm.fingerprint() for key, engine in ENGINES.items()} | |
| _LANGUAGE_NAMES = {e["language_id"]: e["language"] for e in ROSTER} | |
| LANGUAGE_CHOICES = [("Follow the voice", "")] + [ | |
| (f"{_LANGUAGE_NAMES.get(code, code)} ({code})", code) for code in lk.languages() | |
| ] | |
| # -------------------------------------------------------------------------- | |
| # Helpers | |
| # -------------------------------------------------------------------------- | |
| def _sha256_audio(audio: np.ndarray) -> str: | |
| """Hash the waveform, not the file. | |
| `Result.save` appends an unsigned loudkit provenance manifest carrying a | |
| wall-clock creation time, which the library itself calls the one byte range | |
| in which two identical renders may legitimately differ. Hashing the saved | |
| WAV would therefore print two different digests for two identical renders | |
| and read as a determinism failure. The waveform is what the identity | |
| contract makes its promise about, so the waveform is what gets hashed. | |
| """ | |
| return hashlib.sha256(np.ascontiguousarray(audio, dtype=np.float32).tobytes()).hexdigest() | |
| def _write(result: lk.Result, *, voice: str, language: str) -> str: | |
| out = tempfile.NamedTemporaryFile(suffix=".wav", delete=False) | |
| out.close() | |
| # Provenance on: the manifest carries the fingerprint, the recipe and the | |
| # seed, which is the machine-readable marking a synthetic-speech demo should | |
| # be handing out by default. | |
| result.save(out.name, voice=voice, language=language) | |
| return out.name | |
| def _stats(result: lk.Result, model: str) -> str: | |
| seconds = len(result.audio) / result.sample_rate | |
| return ( | |
| f"**{seconds:.1f} s of audio.** {result.timings.describe(seconds)}\n\n" | |
| f"`{model}` · algorithm ID `{result.provenance.algorithm_fingerprint}` · seed `{result.seed}` · " | |
| f"speed `{result.speed:g}x` · {result.sample_rate} Hz" | |
| ) | |
| def _estimate(text: str, *, passes: int = 1, overhead: float = 15.0) -> int: | |
| """Seconds of GPU to ask for. | |
| Speech runs at roughly 14 characters a second, and the render is asked to | |
| keep up with better than real time; the overhead covers the process fork and | |
| the first real CUDA touch. Asking for too much costs queue priority but not | |
| quota, which is charged on effective duration, so this leans generous. | |
| """ | |
| audio_seconds = len((text or "").strip()) / 14.0 | |
| return int(min(180.0, overhead + passes * max(4.0, audio_seconds * 0.9))) | |
| def _check(text: str, limit: int) -> str: | |
| text = (text or "").strip() | |
| if not text: | |
| raise gr.Error("Type something to say.") | |
| if len(text) > limit: | |
| raise gr.Error(f"Keep it under {limit:,} characters here. The library itself takes 10,000.") | |
| return text | |
| def _engine(model: str) -> lk.Engine: | |
| return ENGINES[model if model in ENGINES else DEFAULT_MODEL] | |
| # -------------------------------------------------------------------------- | |
| # Listen. No GPU: these files were rendered ahead of time and ship in the repo. | |
| # -------------------------------------------------------------------------- | |
| def listen(name: str, model: str): | |
| entry = BY_NAME[name] | |
| sample, reference, source = entry["sample"], entry["reference"], entry["source"] | |
| model = model if model in MODELS else DEFAULT_MODEL | |
| lines = [ | |
| f"### {entry['name']}. {entry['language']} ({entry['gender']}).", | |
| "", | |
| f"> {sample['text']}", | |
| "", | |
| f"From *{sample['work']}*, seed `{sample['seed']}`, rendered on `{model}`.", | |
| "", | |
| f"- Reference recording: {reference['duration_s']:.1f} s, {reference['construction']}.", | |
| f"- Source: [{source['name']}]({source['url']}), {source['license']}.", | |
| f"- Consent: {source['consent']}.", | |
| ] | |
| similarity = entry.get("speaker_similarity") | |
| if similarity is not None: | |
| lines.append(f"- Speaker similarity to the reference: {similarity:.3f}.") | |
| lines.append(f"- Voice profile: `{entry['profile']['hf_path']}`, the same file in both models.") | |
| return ( | |
| str(HERE / sample["audio"][model]), | |
| str(HERE / reference["public_preview"]), | |
| "\n".join(lines), | |
| ) | |
| ROSTER_TABLE = [ | |
| [ | |
| entry["name"], | |
| entry["language"], | |
| entry["gender"], | |
| entry["source"]["license"], | |
| f"{entry['speaker_similarity']:.3f}" if entry.get("speaker_similarity") is not None else "", | |
| ] | |
| for entry in ORDERED | |
| ] | |
| # -------------------------------------------------------------------------- | |
| # Speak. GPU. | |
| # -------------------------------------------------------------------------- | |
| def _speak_duration(text, name, model, language, seed, speed): | |
| return _estimate(text, overhead=15.0) | |
| def speak(text: str, name: str, model: str, language: str, seed: float, speed: float): | |
| text = _check(text, MAX_CHARS) | |
| result = _engine(model).synthesize( | |
| text, | |
| PROFILES[name], | |
| seed=int(seed), | |
| language=language or None, | |
| speed=float(speed), | |
| ) | |
| label = language or BY_NAME[name]["language_id"] | |
| return _write(result, voice=name, language=label), _stats(result, model) | |
| # -------------------------------------------------------------------------- | |
| # Both models on one text. GPU. The comparison the model choice is really about. | |
| # -------------------------------------------------------------------------- | |
| def _compare_duration(text, name, seed): | |
| return _estimate(text, passes=2, overhead=20.0) | |
| def compare(text: str, name: str, seed: float): | |
| """The same words, the same voice, the same seed, on both models.""" | |
| text = _check(text, MAX_PROBE_CHARS) | |
| profile = PROFILES[name] | |
| out = {} | |
| for key, engine in ENGINES.items(): | |
| result = engine.synthesize(text, profile, seed=int(seed)) | |
| out[key] = ( | |
| _write(result, voice=name, language=BY_NAME[name]["language_id"]), | |
| len(result.audio) / result.sample_rate, | |
| result.provenance.algorithm_fingerprint, | |
| ) | |
| note = "\n".join( | |
| [ | |
| "| model | audio | algorithm ID |", | |
| "|---|---:|---|", | |
| *( | |
| f"| `{key}` | {seconds:.2f} s | `{fingerprint}` |" | |
| for key, (_, seconds, fingerprint) in out.items() | |
| ), | |
| "", | |
| "Two models, two algorithm IDs, and the same voice profile in both. " | |
| "Which one to ship is an ear question, not a table question.", | |
| ] | |
| ) | |
| return out["loudr-1"][0], out["turbo"][0], note | |
| # -------------------------------------------------------------------------- | |
| # Clone. GPU. The microphone is the default path; an upload is gated. | |
| # -------------------------------------------------------------------------- | |
| def _clone_duration(example, mic, upload, consent, text, model, language, seed, speed): | |
| # Enrollment is a fixed cost on top of the render: two encoders and a | |
| # tokenizer over at most 20 s of audio. | |
| return _estimate(text, overhead=30.0) | |
| def clone(example, mic, upload, consent: bool, text: str, model: str, language: str, seed, speed): | |
| """Enroll a voice, speak with it, keep nothing. | |
| Three ways in, and they do not carry the same consent story, so they are | |
| not collapsed into one input. A shipped example is a clip whose donor | |
| released it for exactly this. A microphone recording is the visitor's own | |
| voice, which is consent by construction. An upload is neither, so it is the | |
| only one gated on a checkbox. | |
| """ | |
| if example: | |
| # A file that ships in this repo. It must survive the request. | |
| source = str(HERE / BY_NAME[example]["reference"]["public_preview"]) | |
| ephemeral = False | |
| label = example | |
| else: | |
| source = mic or upload | |
| ephemeral = True | |
| label = "cloned" | |
| if not source: | |
| raise gr.Error( | |
| "Pick an example, record yourself, or upload a clip you are allowed to use." | |
| ) | |
| if upload and not mic and not consent: | |
| raise gr.Error( | |
| "Confirm the uploaded voice is yours, or that you have permission to use it." | |
| ) | |
| text = _check(text, MAX_CLONE_CHARS) | |
| if example and not language: | |
| language = BY_NAME[example]["language_id"] | |
| try: | |
| import librosa | |
| samples, _ = librosa.load(source, sr=24_000, mono=True) | |
| limit = int(ENROLL_SECONDS * 24_000) | |
| if samples.size > limit: | |
| samples = samples[:limit] | |
| try: | |
| # The profile stays a local. It is never saved, never returned and | |
| # never offered for download: the embeddings are the part of a | |
| # cloned voice that would outlive the request if anything held them. | |
| profile = enroller.enroll(samples, 24_000, name=label) | |
| except ValueError as exc: | |
| # The library's own messages name the bound and describe a good | |
| # input, which is more useful than anything restated here. | |
| raise gr.Error(str(exc)) from exc | |
| # `enroll` writes no language, so every cloned voice would claim English | |
| # and read its text through the English funnel. | |
| profile = dataclasses.replace(profile, language=language or "en") | |
| result = _engine(model).synthesize( | |
| text, profile, seed=int(seed), language=language or None, speed=float(speed) | |
| ) | |
| return _write(result, voice=label, language=profile.language), _stats(result, model) | |
| finally: | |
| # Nothing the visitor recorded outlives the request that carried it. | |
| # A shipped example is not the visitor's and is not ours to delete. | |
| if ephemeral and source: | |
| with contextlib.suppress(OSError): | |
| os.unlink(source) | |
| # -------------------------------------------------------------------------- | |
| # Determinism probe. GPU. Renders the same text twice at the same seed. | |
| # -------------------------------------------------------------------------- | |
| def _probe_duration(text, name, model, seed): | |
| return _estimate(text, passes=2, overhead=20.0) | |
| def probe(text: str, name: str, model: str, seed: float): | |
| text = _check(text, MAX_PROBE_CHARS) | |
| engine = _engine(model) | |
| profile = PROFILES[name] | |
| first = engine.synthesize(text, profile, seed=int(seed)) | |
| second = engine.synthesize(text, profile, seed=int(seed)) | |
| left, right = _sha256_audio(first.audio), _sha256_audio(second.audio) | |
| verdict = "Identical." if left == right else "Different. Please report this." | |
| return "\n".join( | |
| [ | |
| f"**{verdict}**", | |
| "", | |
| "```", | |
| f"render 1 sha256 {left}", | |
| f"render 2 sha256 {right}", | |
| f" algorithm ID {first.provenance.algorithm_fingerprint} seed {int(seed)} {model}", | |
| "```", | |
| "", | |
| "Identical within this build and this device. loudkit promises a " | |
| "bit-identical waveform for the same seed, build, backend and input. " | |
| "It does not promise that your laptop matches this GPU, and the two " | |
| "models are two different builds by construction. " | |
| f"[Read the identity contract]({IDENTITY_CONTRACT}).", | |
| ] | |
| ) | |
| # -------------------------------------------------------------------------- | |
| # Interface | |
| # -------------------------------------------------------------------------- | |
| # loudreader.io: cream ground, ink text, black pill buttons at 14px. | |
| CSS = """ | |
| #lk-head h1 { font-size: 2.15rem; margin-bottom: .25rem; letter-spacing: -.02em; } | |
| #lk-head p { margin-top: 0; } | |
| .lk-card { background: #fffdfa; border: 1px solid #e7e1d7; border-radius: 14px; padding: .35rem 1rem; } | |
| footer { display: none !important; } | |
| """ | |
| # Gradio follows the visitor's system theme unless told otherwise, and this | |
| # palette is light-first. Without this the ink-on-cream tokens below land under | |
| # a dark stylesheet and the text turns near-white on a cream ground. | |
| FORCE_LIGHT = """ | |
| () => { | |
| const url = new URL(window.location); | |
| if (url.searchParams.get('__theme') !== 'light') { | |
| url.searchParams.set('__theme', 'light'); | |
| window.location.replace(url.href); | |
| } | |
| } | |
| """ | |
| THEME = gr.themes.Soft( | |
| primary_hue=gr.themes.colors.gray, | |
| neutral_hue=gr.themes.colors.stone, | |
| font=[gr.themes.GoogleFont("Inter"), "system-ui", "sans-serif"], | |
| ).set( | |
| body_background_fill="#f7f5f2", | |
| body_text_color="#111827", | |
| body_text_color_subdued="#4b5563", | |
| block_background_fill="#fffdfa", | |
| block_border_color="#e7e1d7", | |
| border_color_primary="#e7e1d7", | |
| input_background_fill="#ffffff", | |
| button_primary_background_fill="#111827", | |
| button_primary_background_fill_hover="#374151", | |
| button_primary_text_color="#ffffff", | |
| button_large_radius="14px", | |
| button_small_radius="14px", | |
| ) | |
| with gr.Blocks(title="loudkit", theme=THEME, css=CSS, js=FORCE_LIGHT, fill_width=False) as demo: | |
| gr.Markdown( | |
| f""" | |
| # Twenty-eight voices. Ten languages. Two models. | |
| On-device text to speech, running here on ZeroGPU. | |
| [loudr-1](https://huggingface.co/{MODELS["loudr-1"]["repo"]}) · | |
| [loudr-1-turbo](https://huggingface.co/{MODELS["turbo"]["repo"]}) · | |
| [Code]({DOCS}) · [Responsible use]({RESPONSIBLE_USE}) | |
| Listening costs no GPU. Speaking and cloning spend your daily ZeroGPU quota. | |
| Algorithm IDs: loudr-1 `{FINGERPRINTS["loudr-1"]}`, loudr-1-turbo `{FINGERPRINTS["turbo"]}`. | |
| The same text, voice and seed under one ID give the same audio on the same | |
| device and backend ([identity contract]({IDENTITY_CONTRACT})). | |
| """, | |
| elem_id="lk-head", | |
| ) | |
| with gr.Tabs(): | |
| # ------------- Voices: listen for free, then type your own ------------- | |
| with gr.Tab("Voices"): | |
| gr.Markdown( | |
| "Pick a voice and the sample plays at once. Every voice is here " | |
| "twice, once per model, reading the same passage at the same " | |
| "seed. Those files were rendered ahead of time and use no GPU." | |
| ) | |
| with gr.Row(): | |
| with gr.Column(scale=1): | |
| model_pick = gr.Radio( | |
| MODEL_CHOICES, | |
| value=DEFAULT_MODEL, | |
| label="Model", | |
| info="Switching this changes the sample and what the buttons below run.", | |
| ) | |
| with gr.Row(): | |
| lang_pick = gr.Dropdown( | |
| LANGUAGE_FILTER, value=FIRST_LANGUAGE, label="Language" | |
| ) | |
| pick = gr.Dropdown(FIRST_VOICES, value=FIRST_VOICE, label="Voice") | |
| made = gr.Audio(label="loudkit", type="filepath", interactive=False) | |
| ref = gr.Audio( | |
| label="Reference recording", type="filepath", interactive=False | |
| ) | |
| with gr.Column(scale=1): | |
| card = gr.Markdown(elem_classes="lk-card") | |
| with gr.Accordion("The whole roster", open=False): | |
| gr.Dataframe( | |
| value=ROSTER_TABLE, | |
| headers=["Voice", "Language", "Gender", "Licence", "Similarity"], | |
| interactive=False, | |
| wrap=True, | |
| ) | |
| gr.Markdown("### Say something in this voice.") | |
| gr.Markdown( | |
| f"Up to {MAX_CHARS:,} characters here. The library itself takes 10,000. " | |
| "This part spends your ZeroGPU quota." | |
| ) | |
| say = gr.Textbox( | |
| label="Your text", | |
| value="Hello from loudkit.", | |
| placeholder="Hello from loudkit.", | |
| lines=3, | |
| # Without an explicit ceiling the box renders at its default | |
| # maximum, which is twenty rows of empty space. | |
| max_lines=6, | |
| max_length=MAX_CHARS, | |
| ) | |
| with gr.Row(): | |
| say_lang = gr.Dropdown(LANGUAGE_CHOICES, value="", label="Read the text as") | |
| say_seed = gr.Number(value=7, precision=0, label="Seed") | |
| say_speed = gr.Slider( | |
| lk.MIN_SPEED, lk.MAX_SPEED, value=1.0, step=0.05, label="Speed" | |
| ) | |
| say_go = gr.Button("Speak", variant="primary") | |
| say_out = gr.Audio(label="Speech", type="filepath") | |
| say_stats = gr.Markdown() | |
| say_go.click( | |
| speak, | |
| [say, pick, model_pick, say_lang, say_seed, say_speed], | |
| [say_out, say_stats], | |
| ) | |
| def on_language(language, model): | |
| choices = voices_in(language) | |
| name = choices[0][1] | |
| return (gr.Dropdown(choices=choices, value=name), *listen(name, model)) | |
| lang_pick.change(on_language, [lang_pick, model_pick], [pick, made, ref, card]) | |
| pick.change(listen, [pick, model_pick], [made, ref, card]) | |
| model_pick.change(listen, [pick, model_pick], [made, ref, card]) | |
| demo.load(listen, [pick, model_pick], [made, ref, card]) | |
| with gr.Accordion("Hear both models on your own words", open=False): | |
| gr.Markdown( | |
| "The same text, the same voice, the same seed, rendered on " | |
| "both. This is the comparison the model choice is about, and " | |
| "it costs two renders of your quota." | |
| ) | |
| with gr.Row(): | |
| cmp_text = gr.Textbox( | |
| value="The same words, on both models.", | |
| label="Text", | |
| lines=1, | |
| max_lines=2, | |
| max_length=MAX_PROBE_CHARS, | |
| scale=3, | |
| ) | |
| cmp_seed = gr.Number(value=7, precision=0, label="Seed", scale=1) | |
| cmp_go = gr.Button("Render on both") | |
| with gr.Row(): | |
| cmp_base = gr.Audio(label="loudr-1", type="filepath") | |
| cmp_turbo = gr.Audio(label="loudr-1-turbo", type="filepath") | |
| cmp_note = gr.Markdown() | |
| cmp_go.click(compare, [cmp_text, pick, cmp_seed], [cmp_base, cmp_turbo, cmp_note]) | |
| with gr.Accordion("Determinism check", open=False): | |
| gr.Markdown( | |
| "This renders the same text twice at the same seed on the " | |
| "selected model and hashes both waveforms. The digests must " | |
| "match." | |
| ) | |
| with gr.Row(): | |
| probe_text = gr.Textbox( | |
| value="The same seed gives the same audio.", | |
| label="Text", | |
| lines=1, | |
| max_lines=2, | |
| max_length=MAX_PROBE_CHARS, | |
| scale=3, | |
| ) | |
| probe_seed = gr.Number(value=7, precision=0, label="Seed", scale=1) | |
| probe_go = gr.Button("Render twice") | |
| probe_out = gr.Markdown() | |
| probe_go.click(probe, [probe_text, pick, model_pick, probe_seed], probe_out) | |
| # ---------------------------- Clone ---------------------------- | |
| with gr.Tab("Clone"): | |
| gr.Markdown( | |
| f""" | |
| Clone a voice from a short recording, then speak with it on either model. | |
| - Try one of the shipped examples, or record yourself. | |
| - Clone only your own voice, or a voice you have permission to use. | |
| - One profile serves both models. That is what a portable voice profile means: | |
| enroll once, and the file works wherever the engine does. | |
| - Nothing is kept. The recording and the voice embeddings are discarded when the | |
| request ends, and neither is offered for download. | |
| - See [Responsible use]({RESPONSIBLE_USE}). | |
| """ | |
| ) | |
| with gr.Row(): | |
| with gr.Column(scale=1): | |
| example = gr.Dropdown( | |
| EXAMPLE_CHOICES, | |
| value=CLONE_EXAMPLES[0], | |
| label="Try an example", | |
| info="Reference clips donated for building TTS voices.", | |
| ) | |
| example_ref = gr.Audio( | |
| label="What gets cloned", | |
| type="filepath", | |
| interactive=False, | |
| show_download_button=False, | |
| ) | |
| gr.Markdown("Or use your own voice. That clears the example.") | |
| mic = gr.Audio( | |
| sources=["microphone"], type="filepath", label="Record yourself" | |
| ) | |
| with gr.Accordion("Upload a file instead", open=False): | |
| upload = gr.Audio( | |
| sources=["upload"], type="filepath", label="Audio file" | |
| ) | |
| consent = gr.Checkbox( | |
| value=False, | |
| label=( | |
| "This is my own voice, or I have permission from the " | |
| "person who owns it." | |
| ), | |
| ) | |
| with gr.Column(scale=1): | |
| clone_model = gr.Radio( | |
| MODEL_CHOICES, value=DEFAULT_MODEL, label="Speak it with" | |
| ) | |
| clone_text = gr.Textbox( | |
| label="Text to speak", | |
| value="Now in my own voice.", | |
| placeholder="Now in my own voice.", | |
| lines=3, | |
| max_lines=6, | |
| max_length=MAX_CLONE_CHARS, | |
| ) | |
| clone_lang = gr.Dropdown( | |
| LANGUAGE_CHOICES, value="", label="Language of the text" | |
| ) | |
| with gr.Row(): | |
| clone_seed = gr.Number(value=7, precision=0, label="Seed") | |
| clone_speed = gr.Slider( | |
| lk.MIN_SPEED, lk.MAX_SPEED, value=1.0, step=0.05, label="Speed" | |
| ) | |
| clone_go = gr.Button("Clone and speak", variant="primary") | |
| clone_out = gr.Audio( | |
| label="Speech", type="filepath", show_download_button=False | |
| ) | |
| clone_stats = gr.Markdown() | |
| def show_example(name): | |
| if not name: | |
| return None | |
| return str(HERE / BY_NAME[name]["reference"]["public_preview"]) | |
| def clear_example(value): | |
| # Recording or uploading takes over from the example, so the two | |
| # cannot both be armed and leave the visitor guessing which won. | |
| return gr.Dropdown(value="") if value else gr.skip() | |
| example.change(show_example, example, example_ref) | |
| demo.load(show_example, example, example_ref) | |
| mic.change(clear_example, mic, example) | |
| upload.change(clear_example, upload, example) | |
| clone_go.click( | |
| clone, | |
| [ | |
| example, | |
| mic, | |
| upload, | |
| consent, | |
| clone_text, | |
| clone_model, | |
| clone_lang, | |
| clone_seed, | |
| clone_speed, | |
| ], | |
| [clone_out, clone_stats], | |
| ) | |
| gr.Markdown( | |
| f""" | |
| --- | |
| Run the same engine locally, where nothing is queued and nothing is metered. | |
| ```bash | |
| pip install "loudkit[torch,audio,enroll,hub]" | |
| ``` | |
| ```python | |
| import loudkit as lk | |
| engine = lk.load("{MODELS["loudr-1"]["repo"]}", revision="v0.1.1") | |
| voice = lk.voice("kathleen", repo="{MODELS["loudr-1"]["repo"]}", revision="v0.1.1") | |
| engine.synthesize("Hello from loudkit.", voice, seed=7).save("hello.wav") | |
| ``` | |
| The same profile reads on the faster model by changing one string: | |
| ```python | |
| engine = lk.load("{MODELS["turbo"]["repo"]}", revision="v0.1.1") | |
| ``` | |
| Output files carry [unsigned loudkit provenance metadata]({DOCS}/blob/main/docs/reference/provenance.md): the algorithm ID, the recipe and the seed. | |
| """ | |
| ) | |
| # Each engine holds one set of weights and renders with an internal producer | |
| # thread. One render at a time keeps two requests off the same buffers. | |
| demo.queue(default_concurrency_limit=1, max_size=24) | |
| if __name__ == "__main__": | |
| demo.launch() | |