""" Gradio demo for Tharshan/indicf5_hindi-english_code_switch. Tuned for HF Spaces CPU basic (2 vCPU / 16 GB): - no `spaces` / ZeroGPU decorator (nothing to allocate on CPU) - torch pinned to 2 threads (container over-reports core count) - default 16 denoising steps instead of 32 (~halves inference time) - 150-char cap, validated before any model work - requests queued so each gets the full CPU instead of contending """ import gradio as gr import torch from transformers import AutoModel REPO = "Tharshan/indicf5_hindi-english_code_switch" MAX_CHARS = 150 print("loading model ...") model = AutoModel.from_pretrained(REPO, trust_remote_code=True) model.eval() torch.set_num_threads(2) VOICES = model.voices() CHOICES = [(v.get("name", k), k) for k, v in VOICES.items()] DEFAULT = list(VOICES)[0] print(f"ready — voices: {list(VOICES)}") # Same sentence across languages, so switching the dropdown is directly # comparable. Keys must exist in voices.json or the buttons raise KeyError. _CANDIDATES = [ ("ritu_hinglish", "मैं आज office जा रहा हूँ, morning में meeting है।"), ("ta_hinglish", "நான் இன்னைக்கு office போறேன், morning ல meeting இருக்கு."), ("bn", "আমি আজ office যাচ্ছি, morning এ একটা meeting আছে।"), ("te", "నేను ఈరోజు office కి వెళ్తున్నాను, morning లో meeting ఉంది."), ("kn", "ನಾನು ಇವತ್ತು office ಗೆ ಹೋಗುತ್ತಿದ್ದೇನೆ, morning ನಲ್ಲಿ meeting ಇದೆ."), ("ml", "ഞാൻ ഇന്ന് office ലേക്ക് പോകുന്നു, morning ൽ ഒരു meeting ഉണ്ട്."), ("mr", "मी आज office ला जातोय, morning मध्ये एक meeting आहे."), ("gu", "હું આજે office જઈ રહ્યો છું, morning માં એક meeting છે."), ("pa", "ਮੈਂ ਅੱਜ office ਜਾ ਰਿਹਾ ਹਾਂ, morning ਵਿੱਚ ਇੱਕ meeting ਹੈ।"), ("or", "ମୁଁ ଆଜି office ଯାଉଛି, morning ରେ ଗୋଟିଏ meeting ଅଛି।"), ] # drop any example whose voice isn't actually bundled EXAMPLES = [[text, key] for key, text in _CANDIDATES if key in VOICES] def synth(text, voice_key, nfe, speed): text = (text or "").strip() if not text: raise gr.Error("Enter some text first.") if len(text) > MAX_CHARS: raise gr.Error(f"Keep it under {MAX_CHARS} characters — this runs on " f"free CPU hardware. You entered {len(text)}.") if voice_key not in VOICES: raise gr.Error("Pick a voice from the dropdown.") ref_audio, ref_text = model.voice(voice_key) audio, sr = model.generate(text, ref_audio=ref_audio, ref_text=ref_text, nfe_step=int(nfe), speed=float(speed)) return sr, audio with gr.Blocks(title="Indic code-switched TTS") as demo: gr.Markdown(f""" # Indic code-switched TTS Because nobody says "कार्यालय" when they mean office. Most Indic text-to-speech mangles English words embedded in an Indic sentence. This one doesn't. Fine-tuned from [ai4bharat/IndicF5](https://huggingface.co/ai4bharat/IndicF5) on Hindi-English code-switched speech — the English gain carries across to the other scripts too. > ⏳ Running on free CPU hardware: **30–60 seconds per sentence**, and the first > request after idle is slower while the model wakes up. Keep it short. """) with gr.Row(): with gr.Column(scale=3): text = gr.Textbox( label="Text", lines=2, placeholder="An Indic sentence with English words mixed in…") voice = gr.Dropdown(choices=CHOICES, value=DEFAULT, label="Language / voice") with gr.Accordion("Advanced", open=False): nfe = gr.Slider(8, 32, value=16, step=4, label="Denoising steps — lower is faster, " "quality difference is small") speed = gr.Slider(0.7, 1.3, value=1.0, step=0.05, label="Speed") btn = gr.Button("Generate", variant="primary") with gr.Column(scale=2): out = gr.Audio(label="Output", show_download_button=True) gr.Markdown( "**Tip:** match the voice to your text's language. The reference " "voice sets accent and prosody, and a mismatched one degrades " "output more than any other setting here.") if EXAMPLES: gr.Examples(examples=EXAMPLES, inputs=[text, voice], label="The same sentence across languages") gr.Markdown(f""" --- **Model:** [{REPO}](https://huggingface.co/{REPO}) · fine-tuned from [ai4bharat/IndicF5](https://huggingface.co/ai4bharat/IndicF5) (MIT) on OpenSLR-104 Hindi-English. **Limitations.** Single-pass generation with no text chunking — one or two sentences at a time. Heavy code-switching can drop words. Training audio came from an ASR corpus upsampled to 24 kHz, so timbre is duller than the base model. """) btn.click(synth, inputs=[text, voice, nfe, speed], outputs=out) text.submit(synth, inputs=[text, voice, nfe, speed], outputs=out) demo.queue(max_size=10).launch()