Tharshan's picture
Update app.py
94cc0a6 verified
Raw History Blame Contribute Delete
5.55 kB
"""
Gradio demo for Tharshan/indicf5_hindi-english_code_switch.
Tuned for HF Spaces CPU basic (2 vCPU / 16 GB):
- no `spaces` / ZeroGPU decorator (nothing to allocate on CPU)
- torch pinned to 2 threads (container over-reports core count)
- default 16 denoising steps instead of 32 (~halves inference time)
- 150-char cap, validated before any model work
- requests queued so each gets the full CPU instead of contending
"""
import gradio as gr
import torch
from transformers import AutoModel
REPO = "Tharshan/indicf5_hindi-english_code_switch"
MAX_CHARS = 150
print("loading model ...")
model = AutoModel.from_pretrained(REPO, trust_remote_code=True)
model.eval()
torch.set_num_threads(2)
VOICES = model.voices()
CHOICES = [(v.get("name", k), k) for k, v in VOICES.items()]
DEFAULT = list(VOICES)[0]
print(f"ready — voices: {list(VOICES)}")
# Same sentence across languages, so switching the dropdown is directly
# comparable. Keys must exist in voices.json or the buttons raise KeyError.
_CANDIDATES = [
("ritu_hinglish", "मैं आज office जा रहा हूँ, morning में meeting है।"),
("ta_hinglish", "நான் இன்னைக்கு office போறேன், morning ல meeting இருக்கு."),
("bn", "আমি আজ office যাচ্ছি, morning এ একটা meeting আছে।"),
("te", "నేను ఈరోజు office కి వెళ్తున్నాను, morning లో meeting ఉంది."),
("kn", "ನಾನು ಇವತ್ತು office ಗೆ ಹೋಗುತ್ತಿದ್ದೇನೆ, morning ನಲ್ಲಿ meeting ಇದೆ."),
("ml", "ഞാൻ ഇന്ന് office ലേക്ക് പോകുന്നു, morning ൽ ഒരു meeting ഉണ്ട്."),
("mr", "मी आज office ला जातोय, morning मध्ये एक meeting आहे."),
("gu", "હું આજે office જઈ રહ્યો છું, morning માં એક meeting છે."),
("pa", "ਮੈਂ ਅੱਜ office ਜਾ ਰਿਹਾ ਹਾਂ, morning ਵਿੱਚ ਇੱਕ meeting ਹੈ।"),
("or", "ମୁଁ ଆଜି office ଯାଉଛି, morning ରେ ଗୋଟିଏ meeting ଅଛି।"),
]
# drop any example whose voice isn't actually bundled
EXAMPLES = [[text, key] for key, text in _CANDIDATES if key in VOICES]
def synth(text, voice_key, nfe, speed):
text = (text or "").strip()
if not text:
raise gr.Error("Enter some text first.")
if len(text) > MAX_CHARS:
raise gr.Error(f"Keep it under {MAX_CHARS} characters — this runs on "
f"free CPU hardware. You entered {len(text)}.")
if voice_key not in VOICES:
raise gr.Error("Pick a voice from the dropdown.")
ref_audio, ref_text = model.voice(voice_key)
audio, sr = model.generate(text, ref_audio=ref_audio, ref_text=ref_text,
nfe_step=int(nfe), speed=float(speed))
return sr, audio
with gr.Blocks(title="Indic code-switched TTS") as demo:
gr.Markdown(f"""
# Indic code-switched TTS
Because nobody says "कार्यालय" when they mean office.
Most Indic text-to-speech mangles English words embedded in an Indic sentence.
This one doesn't. Fine-tuned from
[ai4bharat/IndicF5](https://huggingface.co/ai4bharat/IndicF5) on Hindi-English
code-switched speech — the English gain carries across to the other scripts too.
> ⏳ Running on free CPU hardware: **30–60 seconds per sentence**, and the first
> request after idle is slower while the model wakes up. Keep it short.
""")
with gr.Row():
with gr.Column(scale=3):
text = gr.Textbox(
label="Text", lines=2,
placeholder="An Indic sentence with English words mixed in…")
voice = gr.Dropdown(choices=CHOICES, value=DEFAULT,
label="Language / voice")
with gr.Accordion("Advanced", open=False):
nfe = gr.Slider(8, 32, value=16, step=4,
label="Denoising steps — lower is faster, "
"quality difference is small")
speed = gr.Slider(0.7, 1.3, value=1.0, step=0.05, label="Speed")
btn = gr.Button("Generate", variant="primary")
with gr.Column(scale=2):
out = gr.Audio(label="Output", show_download_button=True)
gr.Markdown(
"**Tip:** match the voice to your text's language. The reference "
"voice sets accent and prosody, and a mismatched one degrades "
"output more than any other setting here.")
if EXAMPLES:
gr.Examples(examples=EXAMPLES, inputs=[text, voice],
label="The same sentence across languages")
gr.Markdown(f"""
---
**Model:** [{REPO}](https://huggingface.co/{REPO}) · fine-tuned from
[ai4bharat/IndicF5](https://huggingface.co/ai4bharat/IndicF5) (MIT) on
OpenSLR-104 Hindi-English.
**Limitations.** Single-pass generation with no text chunking — one or two
sentences at a time. Heavy code-switching can drop words. Training audio came
from an ASR corpus upsampled to 24 kHz, so timbre is duller than the base model.
""")
btn.click(synth, inputs=[text, voice, nfe, speed], outputs=out)
text.submit(synth, inputs=[text, voice, nfe, speed], outputs=out)
demo.queue(max_size=10).launch()