Download registry.json from illuspas/Vysengine: direct link, hf CLI and curl.
- Browser
- Download file 43.6 kB
-
https://huggingface.co/illuspas/Vysengine/resolve/main/registry.json
- Command line
-
hf download hf://illuspas/Vysengine/registry.json
-
curl -L -o registry.json https://huggingface.co/illuspas/Vysengine/resolve/main/registry.json
43.6 kB
| { | |
| "version": 1, | |
| "updated_at": "2026-09-24", | |
| "models": [ | |
| { | |
| "id": "qwen3-asr-0.6b", | |
| "name": "Qwen3-ASR 0.6B", | |
| "family": "qwen3_asr", | |
| "task": "asr", | |
| "modes": [ | |
| "offline", | |
| "streaming" | |
| ], | |
| "params": "0.6B", | |
| "languages": [ | |
| "zh", | |
| "en", | |
| "yue", | |
| "ar", | |
| "de", | |
| "fr", | |
| "es", | |
| "pt", | |
| "id", | |
| "it", | |
| "ko", | |
| "ru", | |
| "th", | |
| "vi", | |
| "ja", | |
| "tr", | |
| "hi", | |
| "ms", | |
| "nl", | |
| "sv", | |
| "da", | |
| "fi", | |
| "pl", | |
| "cs", | |
| "fil", | |
| "fa", | |
| "el", | |
| "hu", | |
| "mk", | |
| "ro", | |
| "zh dialects" | |
| ], | |
| "description": "Qwen ASR 0.6B: speech recognition across 30 languages plus 22 Chinese dialects; robust to noise, long audio, and singing.", | |
| "i18n": { | |
| "zh-CN": "Qwen ASR 0.6B, 30 语言识别 + 22 种中文方言, 鲁棒噪声/长音频/歌声。", | |
| "zh-TW": "Qwen ASR 0.6B, 支援 30 種語言辨識 + 22 種中文方言, 對噪聲/長音訊/歌聲具魯棒性。" | |
| }, | |
| "min_engine_version": "0.1.0", | |
| "directory": "Qwen3-ASR-0.6B-GGUF", | |
| "quantizations": [ | |
| { | |
| "name": "q4_k", | |
| "file": "Qwen3-ASR-0.6B-GGUF/qwen3-asr-0.6b-q4_k.gguf", | |
| "size": 930934528, | |
| "sha256": "2493af4978f685c1168b4c414d20e6572184bb6c34948d4de17938e059b20218" | |
| }, | |
| { | |
| "name": "q8_0", | |
| "file": "Qwen3-ASR-0.6B-GGUF/qwen3-asr-0.6b-q8_0.gguf", | |
| "size": 1151275776, | |
| "sha256": "36e55ca79fb39dc57e3ad6827c56e5d7095434c5aa0df1cb03d5aa4d7ccb545b", | |
| "default": true | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "qwen3-asr-1.7b", | |
| "name": "Qwen3-ASR 1.7B", | |
| "family": "qwen3_asr", | |
| "task": "asr", | |
| "modes": [ | |
| "offline", | |
| "streaming" | |
| ], | |
| "params": "1.7B", | |
| "languages": [ | |
| "zh", | |
| "en", | |
| "yue", | |
| "ar", | |
| "de", | |
| "fr", | |
| "es", | |
| "pt", | |
| "id", | |
| "it", | |
| "ko", | |
| "ru", | |
| "th", | |
| "vi", | |
| "ja", | |
| "tr", | |
| "hi", | |
| "ms", | |
| "nl", | |
| "sv", | |
| "da", | |
| "fi", | |
| "pl", | |
| "cs", | |
| "fil", | |
| "fa", | |
| "el", | |
| "hu", | |
| "mk", | |
| "ro", | |
| "zh dialects" | |
| ], | |
| "description": "Qwen ASR 1.7B: speech recognition across 30 languages plus 22 Chinese dialects; robust to noise, long audio, and singing.", | |
| "i18n": { | |
| "zh-CN": "Qwen ASR 1.7B, 30 语言识别 + 22 种中文方言, 鲁棒噪声/长音频/歌声。", | |
| "zh-TW": "Qwen ASR 1.7B, 支援 30 種語言辨識 + 22 種中文方言, 對噪聲/長音訊/歌聲具魯棒性。" | |
| }, | |
| "min_engine_version": "0.1.0", | |
| "directory": "Qwen3-ASR-1.7B-GGUF", | |
| "quantizations": [ | |
| { | |
| "name": "q4_k", | |
| "file": "Qwen3-ASR-1.7B-GGUF/qwen3-asr-1.7b-q4_k.gguf", | |
| "size": 1611869632, | |
| "sha256": "67dee8ca03650f04056f742c33911455361479b0b1c5eb03fd4f4e7239d7e026", | |
| "default": true | |
| }, | |
| { | |
| "name": "q8_0", | |
| "file": "Qwen3-ASR-1.7B-GGUF/qwen3-asr-1.7b-q8_0.gguf", | |
| "size": 2473012672, | |
| "sha256": "663a8122d2635923eebd86795641f7c9d242a8f1c8772c57240cc1b39186c9a3" | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "nemotron-3.5-asr-streaming-0.6b", | |
| "name": "Nemotron 3.5 ASR Streaming 0.6B", | |
| "family": "nemotron_asr", | |
| "task": "asr", | |
| "modes": [ | |
| "offline", | |
| "streaming" | |
| ], | |
| "params": "0.6B", | |
| "languages": [ | |
| "ar-AR", | |
| "bg-BG", | |
| "cs-CZ", | |
| "da-DK", | |
| "de-DE", | |
| "el-GR", | |
| "en-GB", | |
| "en-US", | |
| "es-ES", | |
| "es-US", | |
| "et-EE", | |
| "fi-FI", | |
| "fr-CA", | |
| "fr-FR", | |
| "he-IL", | |
| "hi-IN", | |
| "hr-HR", | |
| "hu-HU", | |
| "it-IT", | |
| "ja-JP", | |
| "ko-KR", | |
| "lt-LT", | |
| "lv-LV", | |
| "mt-MT", | |
| "nb-NO", | |
| "nl-NL", | |
| "nn-NO", | |
| "pl-PL", | |
| "pt-BR", | |
| "pt-PT", | |
| "ro-RO", | |
| "ru-RU", | |
| "sk-SK", | |
| "sl-SI", | |
| "sv-SE", | |
| "th-TH", | |
| "tr-TR", | |
| "uk-UA", | |
| "vi-VN", | |
| "zh-CN" | |
| ], | |
| "description": "NVIDIA Nemotron 3.5 ASR Streaming 0.6B: 600M low-latency streaming speech recognition covering 40 language-region locales, with native punctuation/capitalization, automatic language detection, and configurable chunk size.", | |
| "i18n": { | |
| "zh-CN": "NVIDIA Nemotron 3.5 ASR Streaming 0.6B, 600M 低延迟流式语音识别, 覆盖 40 个语言-地区(locale), 原生标点/大写、自动语言检测, chunk 大小可配置。", | |
| "zh-TW": "NVIDIA Nemotron 3.5 ASR Streaming 0.6B, 600M 低延遲串流語音辨識, 涵蓋 40 個語言-地區(locale), 原生標點/大寫、自動語言偵測, chunk 大小可設定。" | |
| }, | |
| "min_engine_version": "0.1.0", | |
| "directory": "Nemotron-3.5-ASR-Streaming-0.6B-GGUF", | |
| "quantizations": [ | |
| { | |
| "name": "q4_k", | |
| "file": "Nemotron-3.5-ASR-Streaming-0.6B-GGUF/nemotron-3.5-asr-streaming-0.6b-q4_k.gguf", | |
| "size": 694649536, | |
| "sha256": "216e7b1082770c3d4943c0df8ddfe58fac9ee3b8d28dc19c81aab7c36f14f1b4" | |
| }, | |
| { | |
| "name": "q8_0", | |
| "file": "Nemotron-3.5-ASR-Streaming-0.6B-GGUF/nemotron-3.5-asr-streaming-0.6b-q8_0.gguf", | |
| "size": 929756096, | |
| "sha256": "9812d83e24fb1b4838e6861d18b5ebead132e3dc39fdbbeb782ea76d8896b54a", | |
| "default": true | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "qwen3-tts-12hz-1.7b-customvoice", | |
| "name": "Qwen3-TTS 12Hz 1.7B CustomVoice", | |
| "family": "qwen3_tts", | |
| "task": "tts", | |
| "modes": [ | |
| "offline" | |
| ], | |
| "params": "1.7B", | |
| "languages": [ | |
| "zh", | |
| "en", | |
| "ja", | |
| "ko", | |
| "de", | |
| "fr", | |
| "ru", | |
| "pt", | |
| "es", | |
| "it" | |
| ], | |
| "speakers": [ | |
| "serena", | |
| "vivian", | |
| "uncle_fu", | |
| "ryan", | |
| "aiden", | |
| "ono_anna", | |
| "sohee", | |
| "eric", | |
| "dylan" | |
| ], | |
| "description": "Qwen3 TTS 12Hz 1.7B CustomVoice: 9 built-in preset voices controllable via instruction; outputs 24000 Hz mono.", | |
| "i18n": { | |
| "zh-CN": "Qwen3 TTS 12Hz 1.7B CustomVoice, 内置 9 个预设音色, 可经 instruction 控制;输出 24000Hz mono。", | |
| "zh-TW": "Qwen3 TTS 12Hz 1.7B CustomVoice, 內建 9 個預設音色, 可經 instruction 控制;輸出 24000Hz mono。" | |
| }, | |
| "min_engine_version": "0.1.0", | |
| "directory": "Qwen3-TTS-12Hz-1.7B-CustomVoice-GGUF", | |
| "quantizations": [ | |
| { | |
| "name": "q4_k", | |
| "file": "Qwen3-TTS-12Hz-1.7B-CustomVoice-GGUF/qwen3-tts-12hz-1.7b-customvoice-q4_k.gguf", | |
| "size": 2012791008, | |
| "sha256": "444e02414b4ce9aeb800c34995b77fccc732f3fb4dfc340772a371ea7e9a7028" | |
| }, | |
| { | |
| "name": "q8_0", | |
| "file": "Qwen3-TTS-12Hz-1.7B-CustomVoice-GGUF/qwen3-tts-12hz-1.7b-customvoice-q8_0.gguf", | |
| "size": 2817048800, | |
| "sha256": "827f2a3c15902c2bbce7d34ec43feb6eedf0b00ae269602beb214a208d16b93f", | |
| "default": true | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "qwen3-tts-12hz-1.7b-base", | |
| "name": "Qwen3-TTS 12Hz 1.7B Base", | |
| "family": "qwen3_tts", | |
| "task": "tts", | |
| "modes": [ | |
| "offline" | |
| ], | |
| "params": "1.7B", | |
| "languages": [ | |
| "zh", | |
| "en", | |
| "ja", | |
| "ko", | |
| "de", | |
| "fr", | |
| "ru", | |
| "pt", | |
| "es", | |
| "it" | |
| ], | |
| "description": "Qwen3 TTS 12Hz 1.7B Base: requires a voice_ref reference audio for 3-second voice cloning.", | |
| "i18n": { | |
| "zh-CN": "Qwen3 TTS 12Hz 1.7B Base, 需 voice_ref 参考音频做 3 秒声音克隆(voice-clone)。", | |
| "zh-TW": "Qwen3 TTS 12Hz 1.7B Base, 需 voice_ref 參考音訊做 3 秒聲音克隆(voice-clone)。" | |
| }, | |
| "min_engine_version": "0.1.0", | |
| "directory": "Qwen3-TTS-12Hz-1.7B-Base-GGUF", | |
| "quantizations": [ | |
| { | |
| "name": "q4_k", | |
| "file": "Qwen3-TTS-12Hz-1.7B-Base-GGUF/qwen3-tts-12hz-1.7b-base-q4_k.gguf", | |
| "size": 2036805120, | |
| "sha256": "531585471dbc52c651072fe34df1c93190b4bf9adc63937d8a6fb5084dd17202" | |
| }, | |
| { | |
| "name": "q8_0", | |
| "file": "Qwen3-TTS-12Hz-1.7B-Base-GGUF/qwen3-tts-12hz-1.7b-base-q8_0.gguf", | |
| "size": 2841062912, | |
| "sha256": "9b7d4267a66df08e057421975090e09f74454cc3f7952a37ccba8f186bd8eeca", | |
| "default": true | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "qwen3-tts-12hz-0.6b-base", | |
| "name": "Qwen3-TTS 12Hz 0.6B Base", | |
| "family": "qwen3_tts", | |
| "task": "tts", | |
| "modes": [ | |
| "offline" | |
| ], | |
| "params": "0.6B", | |
| "languages": [ | |
| "zh", | |
| "en", | |
| "ja", | |
| "ko", | |
| "de", | |
| "fr", | |
| "ru", | |
| "pt", | |
| "es", | |
| "it" | |
| ], | |
| "description": "Qwen3 TTS 12Hz 0.6B Base: lightweight 0.6B edition; requires a voice_ref reference audio for 3-second voice cloning; outputs 24000 Hz mono.", | |
| "i18n": { | |
| "zh-CN": "Qwen3 TTS 12Hz 0.6B Base, 0.6B 轻量版, 需 voice_ref 参考音频做 3 秒声音克隆(voice-clone);输出 24000Hz mono。", | |
| "zh-TW": "Qwen3 TTS 12Hz 0.6B Base, 0.6B 輕量版, 需 voice_ref 參考音訊做 3 秒聲音克隆(voice-clone);輸出 24000Hz mono。" | |
| }, | |
| "min_engine_version": "0.1.0", | |
| "directory": "Qwen3-TTS-12Hz-0.6B-Base-GGUF", | |
| "quantizations": [ | |
| { | |
| "name": "q4_k", | |
| "file": "Qwen3-TTS-12Hz-0.6B-Base-GGUF/qwen3-tts-12hz-0.6b-base-q4_k.gguf", | |
| "size": 1412004160, | |
| "sha256": "f4c96713ec06f35f2be0a2905fd15e395c20078751dd7396dceb405f228dd1bd" | |
| }, | |
| { | |
| "name": "q8_0", | |
| "file": "Qwen3-TTS-12Hz-0.6B-Base-GGUF/qwen3-tts-12hz-0.6b-base-q8_0.gguf", | |
| "size": 1728149824, | |
| "sha256": "7ed9a05d9c1451de3b114a56be87bdd6604229eb78a2186b36be0de6bf93f8bb", | |
| "default": true | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "qwen3-tts-12hz-0.6b-customvoice", | |
| "name": "Qwen3-TTS 12Hz 0.6B CustomVoice", | |
| "family": "qwen3_tts", | |
| "task": "tts", | |
| "modes": [ | |
| "offline" | |
| ], | |
| "params": "0.6B", | |
| "languages": [ | |
| "zh", | |
| "en", | |
| "ja", | |
| "ko", | |
| "de", | |
| "fr", | |
| "ru", | |
| "pt", | |
| "es", | |
| "it" | |
| ], | |
| "speakers": [ | |
| "serena", | |
| "vivian", | |
| "uncle_fu", | |
| "ryan", | |
| "aiden", | |
| "ono_anna", | |
| "sohee", | |
| "eric", | |
| "dylan" | |
| ], | |
| "description": "Qwen3 TTS 12Hz 0.6B CustomVoice: lightweight 0.6B edition with 9 built-in preset voices controllable via instruction; outputs 24000 Hz mono.", | |
| "i18n": { | |
| "zh-CN": "Qwen3 TTS 12Hz 0.6B CustomVoice, 0.6B 轻量版, 内置 9 个预设音色, 可经 instruction 控制;输出 24000Hz mono。", | |
| "zh-TW": "Qwen3 TTS 12Hz 0.6B CustomVoice, 0.6B 輕量版, 內建 9 個預設音色, 可經 instruction 控制;輸出 24000Hz mono。" | |
| }, | |
| "min_engine_version": "0.1.0", | |
| "directory": "Qwen3-TTS-12Hz-0.6B-CustomVoice-GGUF", | |
| "quantizations": [ | |
| { | |
| "name": "q4_k", | |
| "file": "Qwen3-TTS-12Hz-0.6B-CustomVoice-GGUF/qwen3-tts-12hz-0.6b-customvoice-q4_k.gguf", | |
| "size": 1394283168, | |
| "sha256": "22f2d796b6f6f7136eb4b223151f49598831a83bfd48231d44a97543abba761d" | |
| }, | |
| { | |
| "name": "q8_0", | |
| "file": "Qwen3-TTS-12Hz-0.6B-CustomVoice-GGUF/qwen3-tts-12hz-0.6b-customvoice-q8_0.gguf", | |
| "size": 1710428832, | |
| "sha256": "e207ba7519e8a44bb51e61605ca7f3e8ffa53dbe38dd2626198a6f52b65bab03", | |
| "default": true | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "qwen3-tts-12hz-1.7b-voicedesign", | |
| "name": "Qwen3-TTS 12Hz 1.7B VoiceDesign", | |
| "family": "qwen3_tts", | |
| "task": "vdes", | |
| "modes": [ | |
| "offline" | |
| ], | |
| "params": "1.7B", | |
| "languages": [ | |
| "zh", | |
| "en", | |
| "ja", | |
| "ko", | |
| "de", | |
| "fr", | |
| "ru", | |
| "pt", | |
| "es", | |
| "it" | |
| ], | |
| "description": "Qwen3 TTS 12Hz 1.7B VoiceDesign: generates voices directly from natural-language instruct descriptions without reference audio; outputs 24000 Hz mono.", | |
| "i18n": { | |
| "zh-CN": "Qwen3 TTS 12Hz 1.7B VoiceDesign, 经自然语言 instruct 描述直接生成音色, 无需参考音频;输出 24000Hz mono。", | |
| "zh-TW": "Qwen3 TTS 12Hz 1.7B VoiceDesign, 經自然語言 instruct 描述直接生成音色, 無需參考音訊;輸出 24000Hz mono。" | |
| }, | |
| "min_engine_version": "0.1.0", | |
| "directory": "Qwen3-TTS-12Hz-1.7B-VoiceDesign-GGUF", | |
| "quantizations": [ | |
| { | |
| "name": "q4_k", | |
| "file": "Qwen3-TTS-12Hz-1.7B-VoiceDesign-GGUF/qwen3-tts-12hz-1.7b-voicedesign-q4_k.gguf", | |
| "size": 2012735904, | |
| "sha256": "5e56578c75d11a8f6711af808250fab4ef6f14e3f68f3f80cad661fb1f034eea" | |
| }, | |
| { | |
| "name": "q8_0", | |
| "file": "Qwen3-TTS-12Hz-1.7B-VoiceDesign-GGUF/qwen3-tts-12hz-1.7b-voicedesign-q8_0.gguf", | |
| "size": 2816993696, | |
| "sha256": "2671c0155366bd358a07733b9bde2e67399e6bec24d142387c367bcce2a1ae64", | |
| "default": true | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "fun-asr-nano-2512", | |
| "name": "Fun-ASR-Nano 2512", | |
| "family": "fun_asr_nano", | |
| "task": "asr", | |
| "modes": [ | |
| "offline" | |
| ], | |
| "params": "0.8B", | |
| "languages": [ | |
| "zh", | |
| "en", | |
| "ja" | |
| ], | |
| "description": "FunAudioLLM Fun-ASR-Nano-2512: offline multilingual speech recognition with automatic language detection and ITN (inverse text normalization).", | |
| "i18n": { | |
| "zh-CN": "FunAudioLLM Fun-ASR-Nano-2512 离线多语种语音识别, 支持语言自动检测与 ITN 逆文本归一化。", | |
| "zh-TW": "FunAudioLLM Fun-ASR-Nano-2512 離線多語種語音辨識, 支援語言自動偵測與 ITN 逆文字正規化。" | |
| }, | |
| "min_engine_version": "0.1.0", | |
| "directory": "Fun-ASR-Nano-2512-GGUF", | |
| "quantizations": [ | |
| { | |
| "name": "q4_k", | |
| "file": "Fun-ASR-Nano-2512-GGUF/fun-asr-nano-2512-q4_k.gguf", | |
| "size": 704682176, | |
| "sha256": "ac6191388fe5d8bc77cf217586f02535b700f9efcb5e2caca03c28b8b83bc242" | |
| }, | |
| { | |
| "name": "q8_0", | |
| "file": "Fun-ASR-Nano-2512-GGUF/fun-asr-nano-2512-q8_0.gguf", | |
| "size": 1040881856, | |
| "sha256": "67c829db3e9f2fca962f197aae8428c01787b8b6993d33a8c4e8cd2996af2b52", | |
| "default": true | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "whisper-tiny", | |
| "name": "Whisper Tiny", | |
| "family": "whisper", | |
| "task": "asr", | |
| "modes": [ | |
| "offline" | |
| ], | |
| "params": "39M", | |
| "languages": [ | |
| "zh", | |
| "en", | |
| "yue", | |
| "ar", | |
| "de", | |
| "fr", | |
| "es", | |
| "pt", | |
| "id", | |
| "it", | |
| "ko", | |
| "ru", | |
| "th", | |
| "vi", | |
| "ja", | |
| "tr", | |
| "hi", | |
| "ms", | |
| "nl", | |
| "sv", | |
| "da", | |
| "fi", | |
| "pl", | |
| "cs", | |
| "fil", | |
| "fa", | |
| "el", | |
| "hu", | |
| "mk", | |
| "ro", | |
| "auto" | |
| ], | |
| "description": "OpenAI Whisper Tiny: encoder-decoder multilingual speech recognition with 30-second sliding-window decoding, automatic language detection, and translation.", | |
| "i18n": { | |
| "zh-CN": "OpenAI Whisper Tiny, encoder-decoder 多语种语音识别, 30 秒滑窗解码, 支持语言自动检测与翻译。", | |
| "zh-TW": "OpenAI Whisper Tiny, encoder-decoder 多語種語音辨識, 30 秒滑窗解碼, 支援語言自動偵測與翻譯。" | |
| }, | |
| "min_engine_version": "0.1.0", | |
| "directory": "Whisper-Tiny-GGUF", | |
| "quantizations": [ | |
| { | |
| "name": "f16", | |
| "file": "Whisper-Tiny-GGUF/whisper-tiny-f16.gguf", | |
| "size": 79947488, | |
| "sha256": "f133937a526e3f143d34feee18dd186aea2b23226dcee8dda5947a5ef14f9d5c" | |
| }, | |
| { | |
| "name": "q8_0", | |
| "file": "Whisper-Tiny-GGUF/whisper-tiny-q8_0.gguf", | |
| "size": 65627360, | |
| "sha256": "433727222e486815a5305647bb74934d0fb29c5393fb061c7ed20f934e6cafa9", | |
| "default": true | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "whisper-base", | |
| "name": "Whisper Base", | |
| "family": "whisper", | |
| "task": "asr", | |
| "modes": [ | |
| "offline" | |
| ], | |
| "params": "74M", | |
| "languages": [ | |
| "zh", | |
| "en", | |
| "yue", | |
| "ar", | |
| "de", | |
| "fr", | |
| "es", | |
| "pt", | |
| "id", | |
| "it", | |
| "ko", | |
| "ru", | |
| "th", | |
| "vi", | |
| "ja", | |
| "tr", | |
| "hi", | |
| "ms", | |
| "nl", | |
| "sv", | |
| "da", | |
| "fi", | |
| "pl", | |
| "cs", | |
| "fil", | |
| "fa", | |
| "el", | |
| "hu", | |
| "mk", | |
| "ro", | |
| "auto" | |
| ], | |
| "description": "OpenAI Whisper Base: encoder-decoder multilingual speech recognition with 30-second sliding-window decoding, automatic language detection, and translation.", | |
| "i18n": { | |
| "zh-CN": "OpenAI Whisper Base, encoder-decoder 多语种语音识别, 30 秒滑窗解码, 支持语言自动检测与翻译。", | |
| "zh-TW": "OpenAI Whisper Base, encoder-decoder 多語種語音辨識, 30 秒滑窗解碼, 支援語言自動偵測與翻譯。" | |
| }, | |
| "min_engine_version": "0.1.0", | |
| "directory": "Whisper-Base-GGUF", | |
| "quantizations": [ | |
| { | |
| "name": "q4_k", | |
| "file": "Whisper-Base-GGUF/whisper-base-q4_k.gguf", | |
| "size": 88320736, | |
| "sha256": "7138021fbfbefee5b2786af572c7a58d60ba378d627b751813a1e61b60d73fca" | |
| }, | |
| { | |
| "name": "q8_0", | |
| "file": "Whisper-Base-GGUF/whisper-base-q8_0.gguf", | |
| "size": 110340832, | |
| "sha256": "ed85e994a72e2cdd1602ce6720f835c119fb372135f06b3d86e4243103f6a134", | |
| "default": true | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "whisper-small", | |
| "name": "Whisper Small", | |
| "family": "whisper", | |
| "task": "asr", | |
| "modes": [ | |
| "offline" | |
| ], | |
| "params": "244M", | |
| "languages": [ | |
| "zh", | |
| "en", | |
| "yue", | |
| "ar", | |
| "de", | |
| "fr", | |
| "es", | |
| "pt", | |
| "id", | |
| "it", | |
| "ko", | |
| "ru", | |
| "th", | |
| "vi", | |
| "ja", | |
| "tr", | |
| "hi", | |
| "ms", | |
| "nl", | |
| "sv", | |
| "da", | |
| "fi", | |
| "pl", | |
| "cs", | |
| "fil", | |
| "fa", | |
| "el", | |
| "hu", | |
| "mk", | |
| "ro", | |
| "auto" | |
| ], | |
| "description": "OpenAI Whisper Small: encoder-decoder multilingual speech recognition with 30-second sliding-window decoding, automatic language detection, and translation.", | |
| "i18n": { | |
| "zh-CN": "OpenAI Whisper Small, encoder-decoder 多语种语音识别, 30 秒滑窗解码, 支持语言自动检测与翻译。", | |
| "zh-TW": "OpenAI Whisper Small, encoder-decoder 多語種語音辨識, 30 秒滑窗解碼, 支援語言自動偵測與翻譯。" | |
| }, | |
| "min_engine_version": "0.1.0", | |
| "directory": "Whisper-Small-GGUF", | |
| "quantizations": [ | |
| { | |
| "name": "q4_k", | |
| "file": "Whisper-Small-GGUF/whisper-small-q4_k.gguf", | |
| "size": 207508736, | |
| "sha256": "fb365a10f9331f2725f050cfa703de86c0e31d2e3e126feb83838b458152aba6" | |
| }, | |
| { | |
| "name": "q8_0", | |
| "file": "Whisper-Small-GGUF/whisper-small-q8_0.gguf", | |
| "size": 306599168, | |
| "sha256": "f0a35ba39f7a9085d534e862626d4b1085c3dfe975f1dc0d084b9f76d67097cd", | |
| "default": true | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "whisper-medium", | |
| "name": "Whisper Medium", | |
| "family": "whisper", | |
| "task": "asr", | |
| "modes": [ | |
| "offline" | |
| ], | |
| "params": "769M", | |
| "languages": [ | |
| "zh", | |
| "en", | |
| "yue", | |
| "ar", | |
| "de", | |
| "fr", | |
| "es", | |
| "pt", | |
| "id", | |
| "it", | |
| "ko", | |
| "ru", | |
| "th", | |
| "vi", | |
| "ja", | |
| "tr", | |
| "hi", | |
| "ms", | |
| "nl", | |
| "sv", | |
| "da", | |
| "fi", | |
| "pl", | |
| "cs", | |
| "fil", | |
| "fa", | |
| "el", | |
| "hu", | |
| "mk", | |
| "ro", | |
| "auto" | |
| ], | |
| "description": "OpenAI Whisper Medium: encoder-decoder multilingual speech recognition with 30-second sliding-window decoding, automatic language detection, and translation.", | |
| "i18n": { | |
| "zh-CN": "OpenAI Whisper Medium, encoder-decoder 多语种语音识别, 30 秒滑窗解码, 支持语言自动检测与翻译。", | |
| "zh-TW": "OpenAI Whisper Medium, encoder-decoder 多語種語音辨識, 30 秒滑窗解碼, 支援語言自動偵測與翻譯。" | |
| }, | |
| "min_engine_version": "0.1.0", | |
| "directory": "Whisper-Medium-GGUF", | |
| "quantizations": [ | |
| { | |
| "name": "q4_k", | |
| "file": "Whisper-Medium-GGUF/whisper-medium-q4_k.gguf", | |
| "size": 527553792, | |
| "sha256": "a139a58822db2db1ada6ecec3c170c05c25c2d0508e1e81df89ca54d8607cae2" | |
| }, | |
| { | |
| "name": "q8_0", | |
| "file": "Whisper-Medium-GGUF/whisper-medium-q8_0.gguf", | |
| "size": 879875328, | |
| "sha256": "ef4c5a099b4c0a6d0de19df37a19c9dfea23c3992549a535ad544ef5d3f0a3a5", | |
| "default": true | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "whisper-large", | |
| "name": "Whisper Large", | |
| "family": "whisper", | |
| "task": "asr", | |
| "modes": [ | |
| "offline" | |
| ], | |
| "params": "1.5B", | |
| "languages": [ | |
| "zh", | |
| "en", | |
| "yue", | |
| "ar", | |
| "de", | |
| "fr", | |
| "es", | |
| "pt", | |
| "id", | |
| "it", | |
| "ko", | |
| "ru", | |
| "th", | |
| "vi", | |
| "ja", | |
| "tr", | |
| "hi", | |
| "ms", | |
| "nl", | |
| "sv", | |
| "da", | |
| "fi", | |
| "pl", | |
| "cs", | |
| "fil", | |
| "fa", | |
| "el", | |
| "hu", | |
| "mk", | |
| "ro", | |
| "auto" | |
| ], | |
| "description": "OpenAI Whisper Large: encoder-decoder multilingual speech recognition with 30-second sliding-window decoding, automatic language detection, and translation.", | |
| "i18n": { | |
| "zh-CN": "OpenAI Whisper Large, encoder-decoder 多语种语音识别, 30 秒滑窗解码, 支持语言自动检测与翻译。", | |
| "zh-TW": "OpenAI Whisper Large, encoder-decoder 多語種語音辨識, 30 秒滑窗解碼, 支援語言自動偵測與翻譯。" | |
| }, | |
| "min_engine_version": "0.1.0", | |
| "directory": "Whisper-Large-GGUF", | |
| "quantizations": [ | |
| { | |
| "name": "q4_k", | |
| "file": "Whisper-Large-GGUF/whisper-large-q4_k.gguf", | |
| "size": 993775488, | |
| "sha256": "23953ec179fbb782d21bdad1125dc419f2cc86c76bfc16034f02b3dbcae62fed" | |
| }, | |
| { | |
| "name": "q8_0", | |
| "file": "Whisper-Large-GGUF/whisper-large-q8_0.gguf", | |
| "size": 1727778688, | |
| "sha256": "82046053693322661d78eab600a892a8399467f3c1ec0dae13aa27c22631dc7b", | |
| "default": true | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "whisper-large-v3", | |
| "name": "Whisper Large v3", | |
| "family": "whisper", | |
| "task": "asr", | |
| "modes": [ | |
| "offline" | |
| ], | |
| "params": "1.5B", | |
| "languages": [ | |
| "zh", | |
| "en", | |
| "yue", | |
| "ar", | |
| "de", | |
| "fr", | |
| "es", | |
| "pt", | |
| "id", | |
| "it", | |
| "ko", | |
| "ru", | |
| "th", | |
| "vi", | |
| "ja", | |
| "tr", | |
| "hi", | |
| "ms", | |
| "nl", | |
| "sv", | |
| "da", | |
| "fi", | |
| "pl", | |
| "cs", | |
| "fil", | |
| "fa", | |
| "el", | |
| "hu", | |
| "mk", | |
| "ro", | |
| "auto" | |
| ], | |
| "description": "OpenAI Whisper Large-v3: encoder-decoder multilingual speech recognition with 30-second sliding-window decoding, automatic language detection, and translation.", | |
| "i18n": { | |
| "zh-CN": "OpenAI Whisper Large-v3, encoder-decoder 多语种语音识别, 30 秒滑窗解码, 支持语言自动检测与翻译。", | |
| "zh-TW": "OpenAI Whisper Large-v3, encoder-decoder 多語種語音辨識, 30 秒滑窗解碼, 支援語言自動偵測與翻譯。" | |
| }, | |
| "min_engine_version": "0.1.0", | |
| "directory": "Whisper-Large-V3-GGUF", | |
| "quantizations": [ | |
| { | |
| "name": "q4_k", | |
| "file": "Whisper-Large-V3-GGUF/whisper-large-v3-q4_k.gguf", | |
| "size": 981564352, | |
| "sha256": "5fe9add676f6a250590ff81d20b5b795f2f734062b01157e8ae0a3c15c37cdf5" | |
| }, | |
| { | |
| "name": "q8_0", | |
| "file": "Whisper-Large-V3-GGUF/whisper-large-v3-q8_0.gguf", | |
| "size": 1715567552, | |
| "sha256": "4bbdad671467974e9048d5be6c846f86e96a9f5ace7db2e9ef44f5afcee2136b", | |
| "default": true | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "whisper-large-v3-turbo", | |
| "name": "Whisper Large v3 Turbo", | |
| "family": "whisper", | |
| "task": "asr", | |
| "modes": [ | |
| "offline" | |
| ], | |
| "params": "809M", | |
| "languages": [ | |
| "zh", | |
| "en", | |
| "yue", | |
| "ar", | |
| "de", | |
| "fr", | |
| "es", | |
| "pt", | |
| "id", | |
| "it", | |
| "ko", | |
| "ru", | |
| "th", | |
| "vi", | |
| "ja", | |
| "tr", | |
| "hi", | |
| "ms", | |
| "nl", | |
| "sv", | |
| "da", | |
| "fi", | |
| "pl", | |
| "cs", | |
| "fil", | |
| "fa", | |
| "el", | |
| "hu", | |
| "mk", | |
| "ro", | |
| "auto" | |
| ], | |
| "description": "OpenAI Whisper Large-v3 Turbo: distilled 4-layer decoder variant of encoder-decoder multilingual speech recognition with 30-second sliding-window decoding; roughly 8x faster than large-v3.", | |
| "i18n": { | |
| "zh-CN": "OpenAI Whisper Large-v3 Turbo, decoder 4 层蒸馏版, encoder-decoder 多语种语音识别, 30 秒滑窗解码, 速度约为 large-v3 的 8 倍。", | |
| "zh-TW": "OpenAI Whisper Large-v3 Turbo, decoder 4 層蒸餾版, encoder-decoder 多語種語音辨識, 30 秒滑窗解碼, 速度約為 large-v3 的 8 倍。" | |
| }, | |
| "min_engine_version": "0.1.0", | |
| "directory": "Whisper-Large-V3-Turbo-GGUF", | |
| "quantizations": [ | |
| { | |
| "name": "q4_k", | |
| "file": "Whisper-Large-V3-Turbo-GGUF/whisper-large-v3-turbo-q4_k.gguf", | |
| "size": 567360672, | |
| "sha256": "3769bff47e42b1c296a31bc9eca7ceb1aca82cb09bc4a651a70970d90a7616ed" | |
| }, | |
| { | |
| "name": "q8_0", | |
| "file": "Whisper-Large-V3-Turbo-GGUF/whisper-large-v3-turbo-q8_0.gguf", | |
| "size": 934362272, | |
| "sha256": "74558ebc863364acc4ad896e89c78eff3123a87c0b03478b37be67cdbb3bd044", | |
| "default": true | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "glm-asr-nano-2512", | |
| "name": "GLM-ASR Nano 2512", | |
| "family": "glm_asr", | |
| "task": "asr", | |
| "modes": [ | |
| "offline" | |
| ], | |
| "params": "1.5B", | |
| "languages": [ | |
| "zh", | |
| "en", | |
| "yue", | |
| "auto" | |
| ], | |
| "description": "GLM-ASR-Nano-2512: audio tower + GLM Llama decoder-only architecture; optimized for Chinese, English, and Cantonese dialects, robust to low-volume speech; 30-second windowing with up to ~11 minutes per request.", | |
| "i18n": { | |
| "zh-CN": "GLM-ASR-Nano-2512 语音识别:音频塔 + GLM Llama decoder-only 架构, 中英及粤语等方言优化、低音量语音鲁棒, 30 秒分窗、单次最长约 11 分钟。", | |
| "zh-TW": "GLM-ASR-Nano-2512 語音辨識:音訊塔 + GLM Llama decoder-only 架構, 中英及粵語等方言優化、低音量語音魯棒, 30 秒分窗、單次最長約 11 分鐘。" | |
| }, | |
| "min_engine_version": "0.1.0", | |
| "directory": "GLM-ASR-Nano-2512-GGUF", | |
| "quantizations": [ | |
| { | |
| "name": "q4_k", | |
| "file": "GLM-ASR-Nano-2512-GGUF/glm-asr-nano-2512-q4_k.gguf", | |
| "size": 1460139232, | |
| "sha256": "2774ffbff4b3b579c78f6ab730b80fca0fcd1d580c2344db5617b9704d1a6a00" | |
| }, | |
| { | |
| "name": "q8_0", | |
| "file": "GLM-ASR-Nano-2512-GGUF/glm-asr-nano-2512-q8_0.gguf", | |
| "size": 2525361088, | |
| "sha256": "e524da193bc98558120a0ee5f5e2d45ada3b881c8c6f552b3d3a7584323d8ad7", | |
| "default": true | |
| }, | |
| { | |
| "name": "f16", | |
| "file": "GLM-ASR-Nano-2512-GGUF/glm-asr-nano-2512-f16.gguf", | |
| "size": 4522652608, | |
| "sha256": "1dfcb245c9e080e2ed30acd37af348e787b9417ebf8308099d869032f0842b21" | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "stable-audio-3-small-music", | |
| "name": "Stable Audio 3 Small Music", | |
| "family": "stable_audio", | |
| "task": "gen", | |
| "modes": [ | |
| "offline" | |
| ], | |
| "params": "433M", | |
| "languages": [ | |
| "en" | |
| ], | |
| "description": "Stability AI Stable Audio 3 Small Music: 433M text-to-music with init-audio style rewriting and inpainting, up to 120 seconds; runs on CPU alone; outputs 44100 Hz stereo.", | |
| "i18n": { | |
| "zh-CN": "Stability AI Stable Audio 3 Small Music, 433M 文生音乐, 支持 init-audio 风格改写与 inpainting 局部重绘, 最长 120 秒, CPU 即可运行;输出 44100Hz stereo。", | |
| "zh-TW": "Stability AI Stable Audio 3 Small Music, 433M 文生音樂, 支援 init-audio 風格改寫與 inpainting 局部重繪, 最長 120 秒, CPU 即可執行;輸出 44100Hz stereo。" | |
| }, | |
| "min_engine_version": "0.1.0", | |
| "directory": "Stable-Audio-3-Small-Music-GGUF", | |
| "quantizations": [ | |
| { | |
| "name": "q4_k", | |
| "file": "Stable-Audio-3-Small-Music-GGUF/stable-audio-3-small-music-q4_k.gguf", | |
| "size": 1321224896, | |
| "sha256": "d9647a0b5a7ba663c934b9fac4f32701bc8e169f8ee46bd6b676df734b22dda0" | |
| }, | |
| { | |
| "name": "q8_0", | |
| "file": "Stable-Audio-3-Small-Music-GGUF/stable-audio-3-small-music-q8_0.gguf", | |
| "size": 1683573440, | |
| "sha256": "967b49cb6dba72d008d7623c448e779e27d4a8c9b40eb8075632d32b378d6ba1", | |
| "default": true | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "stable-audio-3-small-sfx", | |
| "name": "Stable Audio 3 Small SFX", | |
| "family": "stable_audio", | |
| "task": "gen", | |
| "modes": [ | |
| "offline" | |
| ], | |
| "params": "433M", | |
| "languages": [ | |
| "en" | |
| ], | |
| "description": "Stability AI Stable Audio 3 Small SFX: 433M text-to-SFX specialist for short sound-effect prompts, up to 120 seconds; runs on CPU alone; outputs 44100 Hz stereo.", | |
| "i18n": { | |
| "zh-CN": "Stability AI Stable Audio 3 Small SFX, 433M 文生音效专用版, 短音效提示词生成, 最长 120 秒, CPU 即可运行;输出 44100Hz stereo。", | |
| "zh-TW": "Stability AI Stable Audio 3 Small SFX, 433M 文生音效專用版, 以短音效提示詞生成, 最長 120 秒, CPU 即可執行;輸出 44100Hz stereo。" | |
| }, | |
| "min_engine_version": "0.1.0", | |
| "directory": "Stable-Audio-3-Small-SFX-GGUF", | |
| "quantizations": [ | |
| { | |
| "name": "q4_k", | |
| "file": "Stable-Audio-3-Small-SFX-GGUF/stable-audio-3-small-sfx-q4_k.gguf", | |
| "size": 1321224832, | |
| "sha256": "9be1baadfe72942c6b0bc57ed1af9fea35d52f7d24cd0c12ac4335e39e1d949c" | |
| }, | |
| { | |
| "name": "q8_0", | |
| "file": "Stable-Audio-3-Small-SFX-GGUF/stable-audio-3-small-sfx-q8_0.gguf", | |
| "size": 1683573376, | |
| "sha256": "23346dc9afa056b365958382420d036ef1e94d63e4cd13e882dca7a6c29ca3f7", | |
| "default": true | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "stable-audio-3-medium", | |
| "name": "Stable Audio 3 Medium", | |
| "family": "stable_audio", | |
| "task": "gen", | |
| "modes": [ | |
| "offline" | |
| ], | |
| "params": "1.4B", | |
| "languages": [ | |
| "en" | |
| ], | |
| "description": "Stability AI Stable Audio 3 Medium: 1.4B high-quality fast generation, text-to-music with init-audio/inpainting, up to 380 seconds; requires GPU; outputs 44100 Hz stereo.", | |
| "i18n": { | |
| "zh-CN": "Stability AI Stable Audio 3 Medium, 1.4B 高质量快速生成, 文生音乐 + init-audio/inpainting, 最长 380 秒, 需 GPU;输出 44100Hz stereo。", | |
| "zh-TW": "Stability AI Stable Audio 3 Medium, 1.4B 高品質快速生成, 文生音樂 + init-audio/inpainting, 最長 380 秒, 需 GPU;輸出 44100Hz stereo。" | |
| }, | |
| "min_engine_version": "0.1.0", | |
| "directory": "Stable-Audio-3-Medium-GGUF", | |
| "quantizations": [ | |
| { | |
| "name": "q4_k", | |
| "file": "Stable-Audio-3-Medium-GGUF/stable-audio-3-medium-q4_k.gguf", | |
| "size": 2378570880, | |
| "sha256": "cc8a2d216fe9e0f700c7d31eb9a4b1316c04a73c6451b9bf21ae0940c12899bf" | |
| }, | |
| { | |
| "name": "q8_0", | |
| "file": "Stable-Audio-3-Medium-GGUF/stable-audio-3-medium-q8_0.gguf", | |
| "size": 3582598272, | |
| "sha256": "8d62175060747a9f5f9e247385becee04a177bc48d8773bff1f5005a0e47f6e1", | |
| "default": true | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "indextts-2.5", | |
| "name": "IndexTTS 2.5", | |
| "family": "index_tts2", | |
| "task": "tts", | |
| "modes": [ | |
| "offline" | |
| ], | |
| "params": "2.3B", | |
| "languages": [ | |
| "zh", | |
| "en", | |
| "ja", | |
| "es", | |
| "ar" | |
| ], | |
| "description": "IndexTeam/bilibili IndexTTS2.5: multilingual zero-shot TTS: 0.8B GPT + DiT CFM + BigVGAN (2.3B full stack); disentangles voice timbre from emotion; requires a voice_ref reference audio for cloning; supports text/audio emotion control and speech-rate control; outputs 22050 Hz mono.", | |
| "i18n": { | |
| "zh-CN": "IndexTeam/bilibili IndexTTS2.5 多语种零样本 TTS:0.8B GPT + DiT CFM + BigVGAN 全栈 2.3B, 音色-情绪解耦, 需 voice_ref 参考音频克隆, 支持文本/音频情绪控制与语速控制;输出 22050Hz mono。", | |
| "zh-TW": "IndexTeam/bilibili IndexTTS2.5 多語種零樣本 TTS:0.8B GPT + DiT CFM + BigVGAN 全棧 2.3B, 音色-情緒解耦, 需 voice_ref 參考音訊克隆, 支援文字/音訊情緒控制與語速控制;輸出 22050Hz mono。" | |
| }, | |
| "min_engine_version": "0.1.0", | |
| "directory": "IndexTTS2.5-GGUF", | |
| "quantizations": [ | |
| { | |
| "name": "q4_k", | |
| "file": "IndexTTS2.5-GGUF/index-tts2_5-q4_k.gguf", | |
| "size": 2758366016, | |
| "sha256": "5acf938702c2d6b9747745fe834dbe495df5ca8d109a8f2007297c039e290d26" | |
| }, | |
| { | |
| "name": "q8_0", | |
| "file": "IndexTTS2.5-GGUF/index-tts2_5-q8_0.gguf", | |
| "size": 3502955328, | |
| "sha256": "244fc74210db31092c6520f6a526e28c7a5adae6a9e28cb9aeff89e3aa518b8f", | |
| "default": true | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "glm-tts", | |
| "name": "GLM-TTS", | |
| "family": "glm_tts", | |
| "task": "tts", | |
| "modes": [ | |
| "offline" | |
| ], | |
| "params": "1.5B", | |
| "languages": [ | |
| "zh", | |
| "en" | |
| ], | |
| "description": "GLM-TTS: zero-shot Chinese/English speech synthesis and voice cloning with full-stack inference (Llama + Whisper-VQ + Flow/DiT + CAMPPlus + HiFT); requires reference audio plus matching reference_text; outputs 24000 Hz mono.", | |
| "i18n": { | |
| "zh-CN": "GLM-TTS 中英零样本语音合成与声音克隆, Llama + Whisper-VQ + Flow/DiT + CAMPPlus + HiFT 全栈推理, 需参考音频及配套参考文本(reference_text);输出 24000Hz mono。", | |
| "zh-TW": "GLM-TTS 中英零樣本語音合成與聲音克隆, Llama + Whisper-VQ + Flow/DiT + CAMPPlus + HiFT 全棧推理, 需參考音訊及配套參考文字(reference_text);輸出 24000Hz mono。" | |
| }, | |
| "min_engine_version": "0.1.0", | |
| "directory": "GLM-TTS-GGUF", | |
| "quantizations": [ | |
| { | |
| "name": "q8_0", | |
| "file": "GLM-TTS-GGUF/glm-tts-q8_0.gguf", | |
| "size": 4190439040, | |
| "sha256": "a9ddfdfd78f0048522e216a6e16b45af464ce723cab9d82bfea6cbb02ce6e790", | |
| "default": true | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "audio8-tts-preview-0.6b", | |
| "name": "Audio8 TTS Preview 0.6B", | |
| "family": "audio8_tts", | |
| "task": "tts", | |
| "modes": [ | |
| "offline" | |
| ], | |
| "params": "0.6B", | |
| "languages": [ | |
| "auto", | |
| "yue", | |
| "zh", | |
| "nl", | |
| "en", | |
| "fr", | |
| "de", | |
| "it", | |
| "ja", | |
| "ko", | |
| "pl", | |
| "es" | |
| ], | |
| "description": "Audio8 TTS Preview 0.6B: three-stage DualAR architecture (slow semantic AR + fast codebook AR + neural codec); zero-shot voice cloning across 12 languages; outputs 44100 Hz mono.", | |
| "i18n": { | |
| "zh-CN": "Audio8 TTS Preview 0.6B, DualAR 三段架构(slow semantic AR + fast codebook AR + 神经 codec), 12 语言零样本声音克隆, 输出 44100Hz mono。", | |
| "zh-TW": "Audio8 TTS Preview 0.6B, DualAR 三段架構(slow semantic AR + fast codebook AR + 神經 codec), 12 語言零樣本聲音克隆, 輸出 44100Hz mono。" | |
| }, | |
| "min_engine_version": "0.1.0", | |
| "directory": "Audio8-TTS-Preview-0.6B-GGUF", | |
| "quantizations": [ | |
| { | |
| "name": "q8_0", | |
| "file": "Audio8-TTS-Preview-0.6B-GGUF/audio8-tts-preview-0.6b-q8_0.gguf", | |
| "size": 1429389792, | |
| "sha256": "4bc9d5828e64d8ff5c4426e2aff33acbf39aeaf229bfce6af0858ddfeaca0e0d", | |
| "default": true | |
| }, | |
| { | |
| "name": "f16", | |
| "file": "Audio8-TTS-Preview-0.6B-GGUF/audio8-tts-preview-0.6b-f16.gguf", | |
| "size": 1890409120, | |
| "sha256": "cc82cf7ff0869da0a4d40cf3682e4900a0a7d715ccbf53fa200112fad8d1cbe6c" | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "moss-tts-local-v1.5", | |
| "name": "MOSS-TTS Local 1.5", | |
| "family": "moss_tts_local", | |
| "task": "tts", | |
| "modes": [ | |
| "offline" | |
| ], | |
| "params": "4.5B", | |
| "languages": [ | |
| "zh", | |
| "yue", | |
| "en", | |
| "ar", | |
| "cs", | |
| "da", | |
| "nl", | |
| "fi", | |
| "fr", | |
| "de", | |
| "el", | |
| "he", | |
| "hi", | |
| "hu", | |
| "it", | |
| "ja", | |
| "ko", | |
| "mk", | |
| "ms", | |
| "fa", | |
| "pl", | |
| "pt", | |
| "ro", | |
| "ru", | |
| "es", | |
| "sw", | |
| "sv", | |
| "tl", | |
| "th", | |
| "tr", | |
| "vi" | |
| ], | |
| "description": "OpenMOSS MOSS-TTS-Local-Transformer-v1.5: Qwen3-4B-class local transformer with MOSS-Audio-Tokenizer v2; zero-shot voice cloning (optional voice_ref) or text-only speech across 31 languages with code-switching, token-level duration control, and inline pause markers; outputs 48000 Hz stereo.", | |
| "i18n": { | |
| "zh-CN": "OpenMOSS MOSS-TTS-Local-Transformer-v1.5, Qwen3-4B 级 local transformer + MOSS-Audio-Tokenizer v2, 31 语言零样本声音克隆(voice_ref 可选)或纯文本合成, 支持码切换、token 级时长控制与 [pause] 停顿标记;输出 48000Hz stereo。", | |
| "zh-TW": "OpenMOSS MOSS-TTS-Local-Transformer-v1.5, Qwen3-4B 級 local transformer + MOSS-Audio-Tokenizer v2, 31 語言零樣本聲音克隆(voice_ref 可選)或純文字合成, 支援碼切換、token 級時長控制與 [pause] 停頓標記;輸出 48000Hz stereo。" | |
| }, | |
| "min_engine_version": "0.2.0", | |
| "directory": "MOSS-TTS-Local-v1.5-GGUF", | |
| "quantizations": [ | |
| { | |
| "name": "q4_k", | |
| "file": "MOSS-TTS-Local-v1.5-GGUF/moss-tts-local-v1.5-q4_k.gguf", | |
| "size": 4389415328, | |
| "sha256": "b91258ea47a7667dc8818014cd0ad2d988089e059a4b2a62fdf7035cd0bab706" | |
| }, | |
| { | |
| "name": "q8_0", | |
| "file": "MOSS-TTS-Local-v1.5-GGUF/moss-tts-local-v1.5-q8_0.gguf", | |
| "size": 7512220768, | |
| "sha256": "ce2d34f73274dcb1e314e884edf6644b74618b99a2ae3256cc4f7b42f59308ea", | |
| "default": true | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "moss-voicegen", | |
| "name": "MOSS VoiceGenerator", | |
| "family": "moss_voicegen", | |
| "task": "vdes", | |
| "modes": [ | |
| "offline" | |
| ], | |
| "params": "1.7B", | |
| "languages": [ | |
| "zh", | |
| "en" | |
| ], | |
| "description": "OpenMOSS MOSS-VoiceGenerator: Qwen3-1.7B delay-pattern backbone (moss_tts_delay) with MOSS-Audio-Tokenizer v1; designs a speaker from a natural-language instruct description without reference audio; English and Chinese; outputs 24000 Hz mono.", | |
| "i18n": { | |
| "zh-CN": "OpenMOSS MOSS-VoiceGenerator, Qwen3-1.7B delay 架构(moss_tts_delay)+ MOSS-Audio-Tokenizer v1, 经自然语言 instruct 描述直接设计音色, 无需参考音频, 支持中英;输出 24000Hz mono。", | |
| "zh-TW": "OpenMOSS MOSS-VoiceGenerator, Qwen3-1.7B delay 架構(moss_tts_delay)+ MOSS-Audio-Tokenizer v1, 經自然語言 instruct 描述直接設計音色, 無需參考音訊, 支援中英;輸出 24000Hz mono。" | |
| }, | |
| "min_engine_version": "0.2.0", | |
| "directory": "MOSS-VoiceGenerator-GGUF", | |
| "quantizations": [ | |
| { | |
| "name": "bf16", | |
| "file": "MOSS-VoiceGenerator-GGUF/moss_voicegen_bf16_codec_f16_decode.gguf", | |
| "size": 6022979936, | |
| "sha256": "5ae9477178606dfc283bb7dcbed1cea0f6a1005d9d35c654f88194100ef7f0fd", | |
| "default": true | |
| } | |
| ] | |
| }, | |
| { | |
| "id": "vibevoice-1.5b", | |
| "name": "VibeVoice 1.5B", | |
| "family": "vibevoice", | |
| "task": "tts", | |
| "modes": [ | |
| "offline" | |
| ], | |
| "params": "1.5B", | |
| "languages": [ | |
| "en", | |
| "zh" | |
| ], | |
| "description": "VibeVoice 1.5B: long-form TTS and multi-speaker dialogue with next-token diffusion; single-speaker voice cloning via voice_ref or up to 4 speakers via voice_samples script prompts; long-form native sliding-window generation; outputs 24000 Hz mono.", | |
| "i18n": { | |
| "zh-CN": "VibeVoice 1.5B, next-token diffusion 长文本 TTS 与多说话人对话生成, voice_ref 单人克隆或 voice_samples 最多 4 说话人脚本, 长文本原生滑窗生成;输出 24000Hz mono。", | |
| "zh-TW": "VibeVoice 1.5B, next-token diffusion 長文字 TTS 與多說話人對話生成, voice_ref 單人克隆或 voice_samples 最多 4 說話人腳本, 長文字原生滑窗生成;輸出 24000Hz mono。" | |
| }, | |
| "min_engine_version": "0.3.0", | |
| "directory": "VibeVoice-1.5B-GGUF", | |
| "quantizations": [ | |
| { | |
| "name": "q4_k", | |
| "file": "VibeVoice-1.5B-GGUF/vibevoice-1.5b-q4_k.gguf", | |
| "size": 2055433602, | |
| "sha256": "d3252f1b57832b28214f50d3e30460562ac56f267645c349e1c53b7f2cff0fd7" | |
| }, | |
| { | |
| "name": "q8_0", | |
| "file": "VibeVoice-1.5B-GGUF/vibevoice-1.5b-q8_0.gguf", | |
| "size": 3224701538, | |
| "sha256": "b6b7cce33bcf656daee03dab8bf94c29eba6f58ffa4481e61e86325466f711f3", | |
| "default": true | |
| }, | |
| { | |
| "name": "bf16", | |
| "file": "VibeVoice-1.5B-GGUF/vibevoice-1.5b-bf16.gguf", | |
| "size": 5420021858, | |
| "sha256": "9f4837d2153e73b4d3154ecab75eb451d5e05d847f56a334db0e1f2774811fff" | |
| } | |
| ] | |
| } | |
| ] | |
| } |