Spaces:
Running on Zero
Running on Zero
File size: 6,866 Bytes
5cb9c90 55282e7 27c943a 5cb9c90 55282e7 5cb9c90 55282e7 5cb9c90 55282e7 5cb9c90 55282e7 5cb9c90 55282e7 5cb9c90 55282e7 5cb9c90 55282e7 5cb9c90 55282e7 5cb9c90 27c943a 5cb9c90 55282e7 5cb9c90 55282e7 5cb9c90 55282e7 5cb9c90 55282e7 5cb9c90 55282e7 5cb9c90 55282e7 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 | # coding=utf-8
import os
import numpy as np
import torch
import torchaudio
import gradio as gr
import spaces
from funasr import AutoModel
model = AutoModel(
model="FunAudioLLM/SenseVoiceSmall",
vad_model="iic/speech_fsmn_vad_zh-cn-16k-common-pytorch",
vad_kwargs={"max_single_segment_time": 30000},
hub="hf",
device="cuda",
)
emo_dict = {
"<|HAPPY|>": "๐", "<|SAD|>": "๐", "<|ANGRY|>": "๐ก",
"<|NEUTRAL|>": "", "<|FEARFUL|>": "๐ฐ", "<|DISGUSTED|>": "๐คข", "<|SURPRISED|>": "๐ฎ",
}
event_dict = {
"<|BGM|>": "๐ผ", "<|Speech|>": "", "<|Applause|>": "๐",
"<|Laughter|>": "๐", "<|Cry|>": "๐ญ", "<|Sneeze|>": "๐คง",
"<|Breath|>": "", "<|Cough|>": "๐ท",
}
emoji_dict = {
"<|nospeech|><|Event_UNK|>": "โ",
"<|zh|>": "", "<|en|>": "", "<|yue|>": "", "<|ja|>": "", "<|ko|>": "",
"<|nospeech|>": "",
"<|HAPPY|>": "๐", "<|SAD|>": "๐", "<|ANGRY|>": "๐ก", "<|NEUTRAL|>": "",
"<|BGM|>": "๐ผ", "<|Speech|>": "", "<|Applause|>": "๐", "<|Laughter|>": "๐",
"<|FEARFUL|>": "๐ฐ", "<|DISGUSTED|>": "๐คข", "<|SURPRISED|>": "๐ฎ",
"<|Cry|>": "๐ญ", "<|EMO_UNKNOWN|>": "", "<|Sneeze|>": "๐คง",
"<|Breath|>": "", "<|Cough|>": "๐ท", "<|Sing|>": "",
"<|Speech_Noise|>": "", "<|withitn|>": "", "<|woitn|>": "",
"<|GBG|>": "", "<|Event_UNK|>": "",
}
lang_dict = {
"<|zh|>": "<|lang|>", "<|en|>": "<|lang|>", "<|yue|>": "<|lang|>",
"<|ja|>": "<|lang|>", "<|ko|>": "<|lang|>", "<|nospeech|>": "<|lang|>",
}
emo_set = {"๐", "๐", "๐ก", "๐ฐ", "๐คข", "๐ฎ"}
event_set = {"๐ผ", "๐", "๐", "๐ญ", "๐คง", "๐ท"}
def format_str_v2(s):
sptk_dict = {}
for sptk in emoji_dict:
sptk_dict[sptk] = s.count(sptk)
s = s.replace(sptk, "")
emo = "<|NEUTRAL|>"
for e in emo_dict:
if sptk_dict.get(e, 0) > sptk_dict.get(emo, 0):
emo = e
for e in event_dict:
if sptk_dict.get(e, 0) > 0:
s = event_dict[e] + s
s = s + emo_dict[emo]
for emoji in emo_set.union(event_set):
s = s.replace(" " + emoji, emoji)
s = s.replace(emoji + " ", emoji)
return s.strip()
def format_str_v3(s):
def get_emo(s):
return s[-1] if s and s[-1] in emo_set else None
def get_event(s):
return s[0] if s and s[0] in event_set else None
s = s.replace("<|nospeech|><|Event_UNK|>", "โ")
for lang in lang_dict:
s = s.replace(lang, "<|lang|>")
s_list = [format_str_v2(s_i).strip(" ") for s_i in s.split("<|lang|>")]
new_s = " " + s_list[0]
cur_ent_event = get_event(new_s)
for i in range(1, len(s_list)):
if len(s_list[i]) == 0:
continue
if get_event(s_list[i]) == cur_ent_event and get_event(s_list[i]) is not None:
s_list[i] = s_list[i][1:]
cur_ent_event = get_event(s_list[i])
if get_emo(s_list[i]) is not None and get_emo(s_list[i]) == get_emo(new_s):
new_s = new_s[:-1]
new_s += s_list[i].strip().lstrip()
new_s = new_s.replace("The.", " ")
return new_s.strip()
@spaces.GPU
def model_inference(input_wav, language, fs=16000):
language = "auto" if not language else language
if isinstance(input_wav, tuple):
fs, input_wav = input_wav
input_wav = input_wav.astype(np.float32) / np.iinfo(np.int16).max
if len(input_wav.shape) > 1:
input_wav = input_wav.mean(-1)
if fs != 16000:
resampler = torchaudio.transforms.Resample(fs, 16000)
input_wav_t = torch.from_numpy(input_wav).to(torch.float32)
input_wav = resampler(input_wav_t[None, :])[0, :].numpy()
text = model.generate(
input=input_wav,
cache={},
language=language,
use_itn=True,
batch_size_s=500,
merge_vad=True,
)
text = text[0]["text"]
text = format_str_v3(text)
return text
audio_examples = [
["example/zh.mp3", "auto"],
["example/en.mp3", "auto"],
["example/yue.mp3", "auto"],
["example/ja.mp3", "auto"],
["example/ko.mp3", "auto"],
["example/emo_1.wav", "auto"],
["example/emo_2.wav", "auto"],
["example/emo_3.wav", "auto"],
["example/rich_1.wav", "auto"],
["example/rich_2.wav", "auto"],
["example/longwav_1.wav", "auto"],
["example/longwav_2.wav", "auto"],
]
description_html = """
<div style="text-align: center; max-width: 800px; margin: 0 auto;">
<h1 style="font-size: 2em; margin-bottom: 0.2em;">๐๏ธ SenseVoice</h1>
<p style="font-size: 1.2em; color: #555; margin-bottom: 0.5em;">Speech Recognition + Emotion Detection + Audio Events โ All in One Model</p>
<p style="font-size: 1em; color: #777;">
<strong>5 languages</strong> (zh/en/yue/ja/ko) ยท <strong>7x faster</strong> than Whisper-small ยท <strong>17x faster</strong> than Whisper-large
</p>
<p style="font-size: 0.9em; margin-top: 1em;">
<a href="https://github.com/FunAudioLLM/SenseVoice" target="_blank">โญ GitHub</a> ยท
<a href="https://github.com/modelscope/FunASR" target="_blank">๐ ๏ธ FunASR Toolkit</a> ยท
<a href="https://arxiv.org/abs/2407.04051" target="_blank">๐ Paper</a> ยท
<a href="https://github.com/FunAudioLLM/Fun-ASR" target="_blank">๐ Fun-ASR (31 Languages)</a>
</p>
</div>
"""
guide_html = """
<div style="background: #f8f9fa; border-radius: 8px; padding: 12px 16px; margin: 8px 0; font-size: 0.9em;">
<strong>How it works:</strong> Upload audio or record via microphone โ SenseVoice transcribes speech and detects emotions (๐๐ก๐) and sound events (๐ผ๐๐๐ญ๐คง).
Event labels appear at the front of text, emotions at the end.
</div>
"""
def launch():
with gr.Blocks(theme=gr.themes.Soft(), title="SenseVoice - Speech Understanding") as demo:
gr.HTML(description_html)
gr.HTML(guide_html)
with gr.Row():
with gr.Column():
audio_inputs = gr.Audio(label="Upload audio or use microphone")
with gr.Accordion("Language (auto-detect by default)", open=False):
language_inputs = gr.Dropdown(
choices=["auto", "zh", "en", "yue", "ja", "ko"],
value="auto",
label="Language",
)
fn_button = gr.Button("Recognize", variant="primary", size="lg")
text_outputs = gr.Textbox(label="Result", lines=5, show_copy_button=True)
gr.Examples(
examples=audio_examples,
inputs=[audio_inputs, language_inputs],
examples_per_page=12,
)
fn_button.click(model_inference, inputs=[audio_inputs, language_inputs], outputs=text_outputs)
demo.launch()
if __name__ == "__main__":
launch()
|