Download app.py from Tharshan/indic-code-switch-tts: direct link, hf CLI and curl.
- Browser
- Download file 5.55 kB
-
https://huggingface.co/spaces/Tharshan/indic-code-switch-tts/resolve/main/app.py
- Command line
-
hf download hf://spaces/Tharshan/indic-code-switch-tts/app.py
-
curl -L -o app.py https://huggingface.co/spaces/Tharshan/indic-code-switch-tts/resolve/main/app.py
5.55 kB
| """ | |
| Gradio demo for Tharshan/indicf5_hindi-english_code_switch. | |
| Tuned for HF Spaces CPU basic (2 vCPU / 16 GB): | |
| - no `spaces` / ZeroGPU decorator (nothing to allocate on CPU) | |
| - torch pinned to 2 threads (container over-reports core count) | |
| - default 16 denoising steps instead of 32 (~halves inference time) | |
| - 150-char cap, validated before any model work | |
| - requests queued so each gets the full CPU instead of contending | |
| """ | |
| import gradio as gr | |
| import torch | |
| from transformers import AutoModel | |
| REPO = "Tharshan/indicf5_hindi-english_code_switch" | |
| MAX_CHARS = 150 | |
| print("loading model ...") | |
| model = AutoModel.from_pretrained(REPO, trust_remote_code=True) | |
| model.eval() | |
| torch.set_num_threads(2) | |
| VOICES = model.voices() | |
| CHOICES = [(v.get("name", k), k) for k, v in VOICES.items()] | |
| DEFAULT = list(VOICES)[0] | |
| print(f"ready — voices: {list(VOICES)}") | |
| # Same sentence across languages, so switching the dropdown is directly | |
| # comparable. Keys must exist in voices.json or the buttons raise KeyError. | |
| _CANDIDATES = [ | |
| ("ritu_hinglish", "मैं आज office जा रहा हूँ, morning में meeting है।"), | |
| ("ta_hinglish", "நான் இன்னைக்கு office போறேன், morning ல meeting இருக்கு."), | |
| ("bn", "আমি আজ office যাচ্ছি, morning এ একটা meeting আছে।"), | |
| ("te", "నేను ఈరోజు office కి వెళ్తున్నాను, morning లో meeting ఉంది."), | |
| ("kn", "ನಾನು ಇವತ್ತು office ಗೆ ಹೋಗುತ್ತಿದ್ದೇನೆ, morning ನಲ್ಲಿ meeting ಇದೆ."), | |
| ("ml", "ഞാൻ ഇന്ന് office ലേക്ക് പോകുന്നു, morning ൽ ഒരു meeting ഉണ്ട്."), | |
| ("mr", "मी आज office ला जातोय, morning मध्ये एक meeting आहे."), | |
| ("gu", "હું આજે office જઈ રહ્યો છું, morning માં એક meeting છે."), | |
| ("pa", "ਮੈਂ ਅੱਜ office ਜਾ ਰਿਹਾ ਹਾਂ, morning ਵਿੱਚ ਇੱਕ meeting ਹੈ।"), | |
| ("or", "ମୁଁ ଆଜି office ଯାଉଛି, morning ରେ ଗୋଟିଏ meeting ଅଛି।"), | |
| ] | |
| # drop any example whose voice isn't actually bundled | |
| EXAMPLES = [[text, key] for key, text in _CANDIDATES if key in VOICES] | |
| def synth(text, voice_key, nfe, speed): | |
| text = (text or "").strip() | |
| if not text: | |
| raise gr.Error("Enter some text first.") | |
| if len(text) > MAX_CHARS: | |
| raise gr.Error(f"Keep it under {MAX_CHARS} characters — this runs on " | |
| f"free CPU hardware. You entered {len(text)}.") | |
| if voice_key not in VOICES: | |
| raise gr.Error("Pick a voice from the dropdown.") | |
| ref_audio, ref_text = model.voice(voice_key) | |
| audio, sr = model.generate(text, ref_audio=ref_audio, ref_text=ref_text, | |
| nfe_step=int(nfe), speed=float(speed)) | |
| return sr, audio | |
| with gr.Blocks(title="Indic code-switched TTS") as demo: | |
| gr.Markdown(f""" | |
| # Indic code-switched TTS | |
| Because nobody says "कार्यालय" when they mean office. | |
| Most Indic text-to-speech mangles English words embedded in an Indic sentence. | |
| This one doesn't. Fine-tuned from | |
| [ai4bharat/IndicF5](https://huggingface.co/ai4bharat/IndicF5) on Hindi-English | |
| code-switched speech — the English gain carries across to the other scripts too. | |
| > ⏳ Running on free CPU hardware: **30–60 seconds per sentence**, and the first | |
| > request after idle is slower while the model wakes up. Keep it short. | |
| """) | |
| with gr.Row(): | |
| with gr.Column(scale=3): | |
| text = gr.Textbox( | |
| label="Text", lines=2, | |
| placeholder="An Indic sentence with English words mixed in…") | |
| voice = gr.Dropdown(choices=CHOICES, value=DEFAULT, | |
| label="Language / voice") | |
| with gr.Accordion("Advanced", open=False): | |
| nfe = gr.Slider(8, 32, value=16, step=4, | |
| label="Denoising steps — lower is faster, " | |
| "quality difference is small") | |
| speed = gr.Slider(0.7, 1.3, value=1.0, step=0.05, label="Speed") | |
| btn = gr.Button("Generate", variant="primary") | |
| with gr.Column(scale=2): | |
| out = gr.Audio(label="Output", show_download_button=True) | |
| gr.Markdown( | |
| "**Tip:** match the voice to your text's language. The reference " | |
| "voice sets accent and prosody, and a mismatched one degrades " | |
| "output more than any other setting here.") | |
| if EXAMPLES: | |
| gr.Examples(examples=EXAMPLES, inputs=[text, voice], | |
| label="The same sentence across languages") | |
| gr.Markdown(f""" | |
| --- | |
| **Model:** [{REPO}](https://huggingface.co/{REPO}) · fine-tuned from | |
| [ai4bharat/IndicF5](https://huggingface.co/ai4bharat/IndicF5) (MIT) on | |
| OpenSLR-104 Hindi-English. | |
| **Limitations.** Single-pass generation with no text chunking — one or two | |
| sentences at a time. Heavy code-switching can drop words. Training audio came | |
| from an ASR corpus upsampled to 24 kHz, so timbre is duller than the base model. | |
| """) | |
| btn.click(synth, inputs=[text, voice, nfe, speed], outputs=out) | |
| text.submit(synth, inputs=[text, voice, nfe, speed], outputs=out) | |
| demo.queue(max_size=10).launch() |