window.ARCHITECTURE = {"schemaVersion":2,"revision":"f825d1d1b92af309585aeb656b2a59c44fc603eb","tasks":[{"id":"speech-synthesis","name":"Speech synthesis"},{"id":"speech-recognition-audio-understanding","name":"Speech recognition & audio understanding"},{"id":"music-sound-video-generation","name":"Music, sound & video generation"},{"id":"voice-conversion-audio-coding","name":"Voice conversion & audio coding"},{"id":"separation-restoration-enhancement","name":"Separation, restoration & enhancement"},{"id":"diarization-activity-detection-alignment","name":"Diarization, activity detection & alignment"},{"id":"built-in-neural-audio-utilities","name":"Built-in neural audio utilities"},{"id":"music-transcription","name":"Music transcription"}],"components":{"ar-qwen3":{"id":"ar-qwen3","name":"Qwen3","role":"ar","summary":"Causal Transformer with RMSNorm, Q/K-normalized attention and SwiGLU. Models share an architecture, not necessarily weights or initialization.","sources":["https://github.com/huggingface/transformers/blob/main/src/transformers/models/qwen3/modeling_qwen3.py"],"diagram":"qwen3-layer"},"codec-higgs-codec":{"id":"codec-higgs-codec","name":"Higgs codec","role":"codec","summary":"Multi-codebook audio reconstruction after undoing the AR delay pattern.","sources":["https://huggingface.co/bosonai/higgs-tts-3-4b"],"diagram":"higgs-codec"},"codec-vocos":{"id":"codec-vocos","name":"Vocos","role":"codec","summary":"ConvNeXt-based spectral synthesis followed by inverse STFT. Acoustic features are supplied by the model.","sources":["https://github.com/gemelo-ai/vocos","https://github.com/ekwek1/soprano"],"diagram":"vocos"},"codec-miocodec":{"id":"codec-miocodec","name":"MioCodec","role":"codec","summary":"Content-token synthesis conditioned on a global voice representation.","sources":["https://github.com/Aratako/MioCodec"],"diagram":"miocodec"},"codec-neucodec":{"id":"codec-neucodec","name":"NeuCodec FSQ decoder","role":"codec","summary":"Finite-scalar token levels feed convolutional residual and rotary Transformer processing, followed by a spectral head and inverse STFT.","sources":["https://huggingface.co/neuphonic/neucodec"],"diagram":"neucodec"},"flow-dit-flow":{"id":"flow-dit-flow","name":"Acoustic patch-flow DiT","role":"flow","summary":"A time-modulated Transformer with convolutional sublayers denoises one continuous patch, conditioned on Qwen3 states and recent patch history. Base additionally supplies CAM++ speaker conditioning.","sources":["https://github.com/FireRedTeam/FireRedTTS3/blob/main/fireredtts3/llm/dit.py"],"diagram":"firered-flow"},"codec-redae":{"id":"codec-redae","name":"RedAE Transformer / ISTFT decoder","role":"codec","summary":"Continuous latents are projected and upsampled, refined by a causal Qwen3-style Transformer, then converted to waveform with a log-magnitude / phase head and ISTFT.","sources":["https://github.com/FireRedTeam/FireRedTTS3/blob/main/fireredtts3/redae/redae.py"],"diagram":"redae-decoder"},"ar-qwen3-talker":{"id":"ar-qwen3-talker","name":"Qwen3 talker","role":"ar","summary":"Temporal AR decoder generates a frame's first codebook and hidden conditioning for the code predictor.","sources":["https://github.com/QwenLM/Qwen3-TTS"],"diagram":"qwen3-layer"},"ar-code-predictor":{"id":"ar-code-predictor","name":"Slot-position acoustic AR","role":"ar","summary":"VieNeu's separate local decoder fills the codebook slots of each frame. It uses learned slot positions rather than RoPE, with a control-token head and individual codebook heads.","sources":["https://github.com/pnnbao97/VieNeu-TTS"],"diagram":"vieneu-depth"},"codec-qwen3-speech-tokenizer":{"id":"codec-qwen3-speech-tokenizer","name":"Qwen3-TTS speech decoder","role":"codec","summary":"Code embeddings pass through a causal Transformer, ConvNeXt rate conversion and SnakeBeta waveform upsampling.","sources":["https://github.com/QwenLM/Qwen3-TTS"],"diagram":"breeze-codec"},"ar-vieneu-talker":{"id":"ar-vieneu-talker","name":"Rotary temporal AR talker","role":"ar","summary":"Trained-from-scratch talker using Qwen3-style RMSNorm, rotary attention with Q/K normalization, and SwiGLU blocks. It is not a pretrained Qwen TTS checkpoint.","sources":["https://huggingface.co/pnnbao-ump/VieNeu-TTS-v3-Turbo"],"diagram":"vieneu-talker"},"codec-moss-audio-tokenizer":{"id":"codec-moss-audio-tokenizer","name":"MOSS causal audio decoder","role":"codec","summary":"Projected codebook embeddings are summed and expanded through causal Transformer stages into waveform samples. Codec versions differ in rate, channels and stage configuration.","sources":["https://huggingface.co/OpenMOSS-Team/MOSS-Audio-Tokenizer"],"diagram":"moss-codec-decoder"},"encoder-t5gemma2":{"id":"encoder-t5gemma2","name":"T5Gemma2 text encoder","role":"encoder","summary":"Gemma text tokens and instruction tags enter a bidirectional encoder with Q/K normalization, rotary attention and gated GELU layers.","sources":["https://huggingface.co/BreezeBlue/Breeze-TTS-2"],"diagram":"breeze-text"},"codec-qwen3-tts-codec":{"id":"codec-qwen3-tts-codec","name":"Qwen3-TTS speech decoder","role":"codec","summary":"Semantic and acoustic code embeddings feed a causal Transformer, ConvNeXt upsampling and a SnakeBeta waveform decoder.","sources":["https://huggingface.co/BreezeBlue/Breeze-TTS-2"],"diagram":"breeze-codec"},"ar-qwen2":{"id":"ar-qwen2","name":"Qwen2","role":"ar","summary":"Qwen2 / Qwen2.5 causal decoder: grouped-query attention and SwiGLU with pre-normalization.","sources":["https://github.com/huggingface/transformers/blob/main/src/transformers/models/qwen2/modeling_qwen2.py"],"diagram":"qwen2-layer"},"codec-hift":{"id":"codec-hift","name":"HiFT vocoder","role":"codec","summary":"Mel-derived pitch generates harmonic excitation; residual upsampling predicts a spectrum for inverse-STFT waveform synthesis.","sources":["https://github.com/FunAudioLLM/CosyVoice","https://github.com/Plachtaa/seed-vc"],"diagram":"hift"},"flow-diffusion-head":{"id":"flow-diffusion-head","name":"Diffusion head","role":"flow","summary":"VibeVoice time-conditioned residual MLP. Not an attention-based DiT.","sources":["https://github.com/microsoft/VibeVoice/blob/main/vibevoice/modular/modular_vibevoice_diffusion_head.py"],"diagram":"vibe-head"},"codec-vibevoice-tokenizer":{"id":"codec-vibevoice-tokenizer","name":"VibeVoice acoustic decoder","role":"codec","summary":"Causal convolutional decoder for continuous acoustic latents.","sources":["https://github.com/microsoft/VibeVoice","https://huggingface.co/microsoft/VibeVoice-1.5B/blob/main/config.json"],"diagram":"vibe-codec"},"flow-soar-meanflow":{"id":"flow-soar-meanflow","name":"SOAR / MeanFlow patch DiT","role":"flow","summary":"An autoregressive flow head generates the next continuous audio patch from Qwen hidden states, acoustic history and optional CAM++ speaker conditioning. MeanFlow also embeds the integration interval.","sources":["https://huggingface.co/dots-studio/dots.tts-soar","https://huggingface.co/dots-studio/dots.tts-mf"],"diagram":"dots-flow"},"codec-dots-audio-vae":{"id":"codec-dots-audio-vae","name":"Dots AudioVAE waveform decoder","role":"codec","summary":"Latent projection and an LSTM feed a BigVGAN-style waveform decoder with transposed convolutions and alias-free SnakeBeta residual blocks.","sources":["https://github.com/studio-dots-ai/dots.tts"],"diagram":"dots-vae-decoder"},"head-convnext":{"id":"head-convnext","name":"Speaker-conditioned ConvNeXt","role":"head","summary":"Speech-token embeddings pass through plain and speaker-modulated ConvNeXt blocks to produce decoder features. The speaker vector is reconstructed from context codes; no diffusion solve or ISTFT occurs here.","sources":["https://github.com/ysharma3501/MiraTTS"],"diagram":"mira-processor"},"codec-snake-conv-decoder":{"id":"codec-snake-conv-decoder","name":"DAC-style waveform decoder","role":"codec","summary":"Convolutional upsampling with Snake activations and dilated residual units produces 16 kHz audio from the acoustic processor's continuous features. FlashSR is a separate final stage.","sources":["https://github.com/ysharma3501/MiraTTS"],"diagram":"mira-decoder"},"ar-llama":{"id":"ar-llama","name":"Llama","role":"ar","summary":"Llama-family causal decoder. Checkpoint lineage and dimensions are model-specific.","sources":["https://github.com/meta-llama/llama/blob/main/llama/model.py"],"diagram":"llama-layer"},"codec-snac":{"id":"codec-snac","name":"SNAC","role":"codec","summary":"Multi-scale discrete audio codec; token streams cover different temporal resolutions.","sources":["https://github.com/hubertsiuzdak/snac","https://huggingface.co/maya-research/maya1"],"diagram":"snac"},"codec-dac":{"id":"codec-dac","name":"DAC waveform decoder","role":"codec","summary":"Projected codebook embeddings are summed and decoded by Snake residual transposed-convolution upsampling. Codebook counts and rates depend on the checkpoint.","sources":["https://github.com/descriptinc/descript-audio-codec"],"diagram":"dac-decoder"},"flow-flow-matching":{"id":"flow-flow-matching","name":"ConvNeXtV2-conditioned flow DiT","role":"flow","summary":"Speech tokens are aligned to mel frames and modeled by ConvNeXtV2 blocks. The rotary DiT uses reference mel and time-plus-speaker adaptive LayerNorm conditioning to predict mel velocity.","sources":["https://github.com/zai-org/GLM-TTS/blob/main/flow/dit.py","https://github.com/zai-org/GLM-TTS/blob/main/flow/modules.py"],"diagram":"glm-flow"},"flow-s3gen":{"id":"flow-s3gen","name":"S3Gen token-to-mel flow","role":"flow","summary":"Speech tokens are encoded and upsampled before a speaker- and reference-conditioned flow network predicts mel features. Turbo uses a distilled MeanFlow time mixer.","sources":["https://github.com/resemble-ai/chatterbox/tree/master/src/chatterbox/models/s3gen"],"diagram":"s3gen"},"ar-gpt-2":{"id":"ar-gpt-2","name":"GPT-2 AR","role":"ar","summary":"Causal Transformer with learned absolute positions, pre-LayerNorm attention and a GELU feed-forward network.","sources":["https://huggingface.co/ResembleAI/chatterbox-turbo"],"diagram":"gpt2-ar"},"head-delayed-codebooks":{"id":"head-delayed-codebooks","name":"Delayed codebook heads","role":"head","summary":"Parallel text/control and audio-codebook heads share an AR hidden state. Staggered codebook positions are realigned before codec decoding; this is not a second depth Transformer.","sources":["https://huggingface.co/OpenMOSS-Team/MOSS-TTS-v1.5","https://huggingface.co/OpenMOSS-Team/MOSS-VoiceGenerator"],"diagram":"moss-delay-heads"},"ar-local-transformer":{"id":"ar-local-transformer","name":"Rotary depth AR","role":"ar","summary":"MOSS Local's small causal Transformer expands one Qwen3 state into a frame of codebooks. A binary head controls continuation; the depth layers use LayerNorm and SiLU feed-forward.","sources":["https://huggingface.co/OpenMOSS-Team/MOSS-TTS-Local-Transformer-v1.5"],"diagram":"moss-local-depth"},"ar-global-transformer":{"id":"ar-global-transformer","name":"Nano temporal AR Transformer","role":"ar","summary":"MOSS Nano uses its own LayerNorm / rotary / GELU Transformer over text and summed audio-frame embeddings, rather than a pretrained Qwen backbone.","sources":["https://huggingface.co/OpenMOSS-Team/MOSS-TTS-Nano-100M"],"diagram":"moss-nano-global"},"codec-fish-dac":{"id":"codec-fish-dac","name":"Fish DAC audio decoder","role":"codec","summary":"Projected semantic and residual codebook entries are summed, refined by a windowed Transformer and ConvNeXt upsampling, then decoded by a Snake waveform CNN.","sources":["https://github.com/fishaudio/fish-speech"],"diagram":"fish-dac-codes"},"codec-transformer-convnext":{"id":"codec-transformer-convnext","name":"Audio8 RVQ waveform decoder","role":"codec","summary":"Codebook lookup feeds a windowed Transformer, ConvNeXt rate conversion and Snake waveform decoder. Codec dimensions and weights belong to Audio8, not Fish S2 Pro.","sources":["https://github.com/Audio8-AI/Audio8_TTS"],"diagram":"audio8-codec"},"ar-flowlm":{"id":"ar-flowlm","name":"FlowLM latent AR","role":"ar","summary":"A causal rotary Transformer conditions each acoustic step on text, voice prompt and previous continuous latents. Its hidden state drives a separate flow MLP and stop head.","sources":["https://github.com/kyutai-labs/pocket-tts"],"diagram":"pocket-flowlm"},"flow-flow-matching-head":{"id":"flow-flow-matching-head","name":"Per-frame flow MLP","role":"flow","summary":"An adaptive-normalized residual MLP uses the AR hidden state and start/end times to transform noise into the next continuous audio latent.","sources":["https://github.com/kyutai-labs/pocket-tts"],"diagram":"pocket-flow-head"},"codec-mimi":{"id":"codec-mimi","name":"Mimi quantized speech decoder","role":"codec","summary":"Semantic and acoustic codebook lookups reconstruct a latent, then rate upsampling, causal rotary attention and an ELU convolutional decoder produce speech.","sources":["https://github.com/kyutai-labs/moshi"],"diagram":"mimi-decoder"},"codec-voxcpm-audiovae":{"id":"codec-voxcpm-audiovae","name":"AudioVAE waveform decoder","role":"codec","summary":"Continuous latents feed causal convolutional upsampling and Snake residual blocks. V1 and V2 use separately trained AudioVAEs; V2 encodes at 16 kHz and synthesizes at 48 kHz.","sources":["https://huggingface.co/openbmb/VoxCPM2","https://huggingface.co/openbmb/VoxCPM-0.5B"],"diagram":"voxcpm-vae-decoder"},"ar-semantic-transformer":{"id":"ar-semantic-transformer","name":"Sopro semantic AR Transformer","role":"ar","summary":"Text, attention-pooled reference style tokens and a semantic continuation prefix condition a causal rotary Transformer that predicts semantic tokens.","sources":["https://github.com/samuel-vitorino/sopro"],"diagram":"sopro-ar"},"flow-acoustic-dit":{"id":"flow-acoustic-dit","name":"Sopro acoustic DiT flow","role":"flow","summary":"Upsampled semantic features, speaker/style conditioning, reference mel and noisy mel feed a time-conditioned diffusion Transformer. Iterative flow integration produces mel features for Vocos.","sources":["https://github.com/samuel-vitorino/sopro"],"diagram":"sopro-flow"},"encoder-text-convnext":{"id":"encoder-text-convnext","name":"ConvNeXt text encoder","role":"encoder","summary":"Character embeddings and positional features pass through ConvNeXt V2 blocks with global-response normalization, padded to the acoustic sequence length.","sources":["https://github.com/SWivid/F5-TTS"],"diagram":"f5-text"},"encoder-zipformer":{"id":"encoder-zipformer","name":"Zipformer","role":"encoder","summary":"Multi-rate encoder stacks combine shared attention weights, nonlinear attention, convolution and learned bypass paths.","sources":["https://huggingface.co/Banafo/Kroko-ASR","https://github.com/k2-fsa/icefall"],"diagram":"zipformer"},"flow-zipformer":{"id":"flow-zipformer","name":"Zipformer acoustic flow","role":"flow","summary":"Multi-rate Zipformer stacks predict mel velocity conditioned on expanded text features, reference mel and time. This is not AR token generation.","sources":["https://github.com/k2-fsa/ZipVoice"],"diagram":"zipvoice-flow"},"flow-qwen3-diffusion":{"id":"flow-qwen3-diffusion","name":"Qwen3 discrete diffusion","role":"flow","summary":"Non-autoregressive Qwen3 blocks iteratively fill masked audio-code positions. Summed codebook embeddings and text/style conditioning feed parallel codebook heads; confidence-based selection progressively accepts predictions.","sources":["https://github.com/k2-fsa/OmniVoice/blob/main/omnivoice/models/omnivoice.py"],"diagram":"omnivoice-diffusion"},"codec-higgs-audio-v2":{"id":"codec-higgs-audio-v2","name":"Higgs Audio V2 decoder","role":"codec","summary":"Residual codebook embeddings are summed and projected into a Snake residual convolutional waveform decoder.","sources":["https://github.com/k2-fsa/OmniVoice/blob/main/omnivoice/models/omnivoice.py"],"diagram":"higgs-v2-decoder"},"encoder-text-caption-encoders":{"id":"encoder-text-caption-encoders","name":"Rotary text / caption Transformer","role":"encoder","summary":"Irodori v3 uses learned token embeddings and bidirectional gated rotary Transformer blocks for text and, in VoiceDesign, captions.","sources":["https://github.com/Aratako/Irodori-TTS"],"diagram":"irodori-condition"},"flow-rf-dit":{"id":"flow-rf-dit","name":"Rectified-flow DiT","role":"flow","summary":"Time-adaptive RMS normalization and gated joint attention condition noisy audio latents on text, reference and optional caption states.","sources":["https://github.com/Aratako/Irodori-TTS"],"diagram":"irodori-flow"},"codec-dac-vae":{"id":"codec-dac-vae","name":"DAC-VAE waveform decoder","role":"codec","summary":"Continuous latents are projected into a Snake residual convolutional decoder. Unlike DAC token decoding, this path does not select codebook entries.","sources":["https://github.com/Aratako/Irodori-TTS"],"diagram":"irodori-codec"},"encoder-modernbert":{"id":"encoder-modernbert","name":"ModernBERT-JA encoder","role":"encoder","summary":"Irodori v4 uses the Japanese ModernBERT backbone with alternating local/global rotary attention and separate text/caption projectors.","sources":["https://huggingface.co/Aratako/Irodori-TTS-v4.1-Small"],"diagram":"irodori-modernbert"},"encoder-gemma3":{"id":"encoder-gemma3","name":"Gemma3 prompt encoder","role":"encoder","summary":"A causal Gemma3 language backbone supplies normalized, projected hidden states from its layers as prompt features. It is used for conditioning, not autoregressive audio generation.","sources":["https://huggingface.co/ResembleAI/Dramabox"],"diagram":"dramabox-gemma"},"codec-bigvgan":{"id":"codec-bigvgan","name":"BigVGAN vocoder","role":"codec","summary":"Mel-conditioned transposed-convolution upsampling with alias-controlled periodic activations and multi-receptive-field residual blocks produces waveform samples.","sources":["https://github.com/NVIDIA/BigVGAN"],"diagram":"bigvgan"},"encoder-pl-bert":{"id":"encoder-pl-bert","name":"PL-BERT / ALBERT","role":"encoder","summary":"Bidirectional phoneme Transformer with factorized embeddings and shared ALBERT layers; contextual features drive duration and prosody.","sources":["https://github.com/yl4579/PL-BERT","https://github.com/hexgrad/kokoro"],"diagram":"plbert"},"ar-gpt":{"id":"ar-gpt","name":"GPT-style semantic AR","role":"ar","summary":"IndexTTS causal Transformer with learned positions, LayerNorm and GELU. Version-specific speaker, emotion and duration prompts condition semantic generation.","sources":["https://github.com/index-tts/index-tts"],"diagram":"index-gpt"},"flow-s2mel":{"id":"flow-s2mel","name":"DiT + WaveNet acoustic flow","role":"flow","summary":"Length-regulated semantic features, reference mel and CAM++ style condition an adaptive rotary DiT with convolutional refinement.","sources":["https://github.com/index-tts/index-tts"],"diagram":"index-s2mel"},"ar-t2s-transformer":{"id":"ar-t2s-transformer","name":"GPT-style semantic AR","role":"ar","summary":"Learned absolute positions, LayerNorm, causal multi-head attention and GELU feed-forward layers predict semantic codes. Hidden states also condition S2A.","sources":["https://github.com/netease-youdao/Confucius4-TTS/blob/main/confuciustts/llm/llm.py"],"diagram":"confucius-t2s"},"flow-s2a":{"id":"flow-s2a","name":"DiT + WaveNet acoustic flow","role":"flow","summary":"Semantic code embeddings and AR hidden states are aligned to mel rate. A rotary DiT with long skips and gated convolutional refinement predicts mel velocity.","sources":["https://github.com/netease-youdao/Confucius4-TTS/blob/main/confuciustts/flow/DiT/dit.py"],"diagram":"confucius-s2a"},"codec-nemo-audio-codec":{"id":"codec-nemo-audio-codec","name":"NeMo NanoCodec","role":"codec","summary":"Finite scalar code indices reconstruct scalar latent groups, then a causal convolutional decoder upsamples them to waveform audio.","sources":["https://huggingface.co/nvidia/magpie_tts_multilingual_357m"],"diagram":"nemo-nanocodec"},"encoder-whisper":{"id":"encoder-whisper","name":"Whisper encoder","role":"encoder","summary":"Convolutional mel frontend and non-causal Transformer layers. These models use the audio encoder, not Whisper's text decoder; width and pooling are checkpoint-specific.","sources":["https://huggingface.co/bosonai/higgs-audio-v3-stt","https://huggingface.co/OpenMOSS-Team/MOSS-Transcribe-Diarize/blob/main/config.json","https://github.com/SamsungLabs/samsone","https://github.com/SamsungLabs/samsone/blob/main/configs/Samsone134M.yaml","https://github.com/Plachtaa/seed-vc"],"diagram":"whisper-encoder"},"encoder-sensevoice-sanm":{"id":"encoder-sensevoice-sanm","name":"SenseVoice / SANM","role":"encoder","summary":"Self-attention combined with feed-forward sequential memory filters. This is not a Conformer convolution block.","sources":["https://github.com/FunAudioLLM/SenseVoice","https://huggingface.co/FunAudioLLM/Fun-ASR-Nano-2512/blob/main/config.yaml"],"diagram":"sanm"},"encoder-qwen3-asr-encoder":{"id":"encoder-qwen3-asr-encoder","name":"Qwen3-ASR audio encoder","role":"encoder","summary":"Strided convolutional frontend and windowed audio Transformer, projected to text-decoder width.","sources":["https://github.com/QwenLM/Qwen3-ASR/blob/main/qwen_asr/core/transformers_backend/modeling_qwen3_asr.py"],"diagram":"qwen-audio"},"encoder-vibevoice-tokenizers":{"id":"encoder-vibevoice-tokenizers","name":"Dual acoustic / semantic encoders","role":"encoder","summary":"Parallel causal convolutional waveform encoders with residual depthwise-convolution blocks. Separate connectors map features to LM width before elementwise addition.","sources":["https://github.com/microsoft/VibeVoice","https://huggingface.co/microsoft/VibeVoice-ASR"],"diagram":"vibe-asr-encoders"},"encoder-vibeasr-vae-encoder":{"id":"encoder-vibeasr-vae-encoder","name":"Dual quantized waveform encoders","role":"encoder","summary":"VibeASR's acoustic and semantic encoders use causal convolutions, RMSNorm and ReLU MLP blocks in the published I8_S package.","sources":["https://github.com/microsoft/VibeASR.cpp"],"diagram":"vibeasr-encoders"},"ar-bitnet-language-model":{"id":"ar-bitnet-language-model","name":"Qwen2 / BitNet","role":"ar","summary":"Qwen2-family causal decoder topology with ternary I2_S and Q6_K projections in VibeASR. Quantized storage is not a different attention architecture.","sources":["https://huggingface.co/microsoft/VibeVoice-ASR-BitNet","https://github.com/microsoft/VibeASR.cpp"],"diagram":"qwen2-layer"},"encoder-voxtral-audio-encoder":{"id":"encoder-voxtral-audio-encoder","name":"Causal audio Transformer","role":"encoder","summary":"Voxtral's convolutional frontend and rotary sliding-window attention produce grouped audio features for a GELU projection adapter.","sources":["https://huggingface.co/mistralai/Voxtral-Mini-4B-Realtime-2602"],"diagram":"voxtral-encoder"},"ar-voxtral-decoder":{"id":"ar-voxtral-decoder","name":"Delay-conditioned AR Transformer","role":"ar","summary":"Projected acoustic frames are added to text embeddings. RMSNorm, rotary causal attention and delay-conditioned feed-forward modulation support streaming transcription.","sources":["https://huggingface.co/mistralai/Voxtral-Mini-4B-Realtime-2602"],"diagram":"voxtral-decoder"},"encoder-pooled-mlp-adaptor":{"id":"encoder-pooled-mlp-adaptor","name":"Pooled residual MLP adapter","role":"encoder","summary":"SAMSONE pools audio features and uses a residual two-layer projection with LayerNorm.","sources":["https://github.com/SamsungLabs/samsone","https://github.com/SamsungLabs/samsone/blob/main/configs/Samsone134M.yaml"],"diagram":"samsone-projector"},"ar-smollm2":{"id":"ar-smollm2","name":"SmolLM2","role":"ar","summary":"Compact Llama-style causal decoder. SAMSONE prunes / distills this backbone and uses a remapped vocabulary; it does not share Maya1 weights.","sources":["https://github.com/SamsungLabs/samsone","https://huggingface.co/HuggingFaceTB/SmolLM2-135M"],"diagram":"smollm2-layer"},"encoder-fastconformer":{"id":"encoder-fastconformer","name":"FastConformer","role":"encoder","summary":"Convolutional subsampling and Macaron-style Conformer blocks. Context and streaming cache policies remain model-specific.","sources":["https://huggingface.co/nvidia/canary-180m-flash","https://huggingface.co/nvidia/nemotron-3.5-asr-streaming-0.6b","https://huggingface.co/nvidia/parakeet-tdt-0.6b-v3","https://huggingface.co/nvidia/diar_sortformer_4spk-v1","https://huggingface.co/nvidia/diar_streaming_sortformer_4spk-v2.1"],"diagram":"fastconformer"},"head-rnn-t":{"id":"head-rnn-t","name":"RNN-T","role":"head","summary":"Recurrent token predictor and acoustic-text joint network.","sources":["https://huggingface.co/nvidia/nemotron-3.5-asr-streaming-0.6b","https://huggingface.co/ai-sage/GigaAM-v3","https://huggingface.co/ai-sage/GigaAM-Multilingual"],"diagram":"rnnt"},"head-tdt":{"id":"head-tdt","name":"TDT","role":"head","summary":"Transducer predicts tokens and durations, allowing larger jumps through acoustic frames.","sources":["https://huggingface.co/nvidia/parakeet-tdt-0.6b-v3"],"diagram":"tdt"},"head-ctc":{"id":"head-ctc","name":"CTC","role":"head","summary":"Framewise label probabilities followed by blank removal and repeated-label collapse.","sources":["https://huggingface.co/ai-sage/GigaAM-v3","https://huggingface.co/ai-sage/GigaAM-Multilingual","https://huggingface.co/ibm-granite/granite-speech-5.0-470m-turboctc","https://huggingface.co/nvidia/stt_en_citrinet_256_ls","https://huggingface.co/FunAudioLLM/SenseVoiceSmall","https://huggingface.co/abr-ai/niagara-19m-batch.en","https://huggingface.co/MahmoudAshraf/mms-300m-1130-forced-aligner/blob/main/config.json"],"diagram":"ctc"},"encoder-citrinet":{"id":"encoder-citrinet","name":"Citrinet","role":"encoder","summary":"Residual time-channel separable convolution blocks with squeeze-and-excitation.","sources":["https://huggingface.co/nvidia/stt_en_citrinet_256_ls"],"diagram":"citrinet"},"encoder-niagara-ssm-attention":{"id":"encoder-niagara-ssm-attention","name":"Niagara SSM + attention","role":"encoder","summary":"Relative attention and state-space temporal filters between feed-forward sublayers. Not a Mamba encoder.","sources":["https://huggingface.co/abr-ai/niagara-19m-batch.en"],"diagram":"niagara"},"encoder-qwen2-5-omni":{"id":"encoder-qwen2-5-omni","name":"Qwen2.5-Omni conditioning","role":"encoder","summary":"Text tokens and optional audio features enter the Qwen2.5-Omni thinker. A learned mixture of layer states conditions generation; the thinker does not autoregressively produce the output waveform.","sources":["https://github.com/Tencent-Hunyuan/AuK"],"diagram":"auk-conditioning"},"flow-conditional-transformer":{"id":"flow-conditional-transformer","name":"Dual / single-stream flow DiT","role":"flow","summary":"Joint audio-text attention first keeps separate stream projections, then merges the streams. Time-conditioned adaptive normalization and SwiGLU layers predict latent velocity.","sources":["https://github.com/Tencent-Hunyuan/AuK/blob/main/src/auk/model/flux2_edit.py"],"diagram":"auk-flow"},"codec-convolutional-vae":{"id":"codec-convolutional-vae","name":"BigVGAN-style VAE decoder","role":"codec","summary":"Continuous audio latents are decoded with transposed convolutions and alias-free SnakeBeta residual blocks into a mono waveform.","sources":["https://github.com/Tencent-Hunyuan/AuK/blob/main/src/auk/model/vae/bigvgan_flow_vae.py"],"diagram":"auk-vae-decoder"},"ar-qwen3-5":{"id":"ar-qwen3-5","name":"Qwen3.5 hybrid AR","role":"ar","summary":"Hybrid recurrent Gated DeltaNet and full-attention layers with SwiGLU feed-forward branches. Distinct from Qwen3's attention-only architecture.","sources":["https://huggingface.co/FireRedTeam/FireRedAudio","https://github.com/huggingface/transformers/blob/main/src/transformers/models/qwen3_5/modeling_qwen3_5.py"],"diagram":"qwen35"},"ar-moshi-depformer":{"id":"ar-moshi-depformer","name":"Moshi depth AR Transformer","role":"ar","summary":"A smaller Transformer predicts audio codebooks sequentially within each frame. It conditions on the temporal state and text/previous codebook tokens, with codebook-specific weights and no rotary positions.","sources":["https://github.com/NVIDIA/personaplex"],"diagram":"moshi-depth"},"ar-mixture-of-transformers":{"id":"ar-mixture-of-transformers","name":"YuE2 AR Transformer","role":"ar","summary":"Causal RMSNorm, Q/K-normalized rotary grouped-query attention and SwiGLU. Separate NAR parameters share the model but not this token-generation path.","sources":["https://github.com/multimodal-art-projection/YuE/blob/main/src/yue2/modeling_yue2.py"],"diagram":"yue2-ar"},"flow-mixture-of-transformers":{"id":"flow-mixture-of-transformers","name":"YuE2 NAR flow Transformer","role":"flow","summary":"Acoustic queries attend to the cached AR prefix and all acoustic positions using NAR-specific projections and MLPs. Time and latent-position embeddings condition the flow solve.","sources":["https://github.com/multimodal-art-projection/YuE/blob/main/src/yue2/modeling_yue2.py"],"diagram":"yue2-nar"},"codec-heartcodec":{"id":"codec-heartcodec","name":"HeartCodec flow + scalar decoder","role":"codec","summary":"Summed quantizer embeddings condition a two-stage flow Transformer. Reconstructed scalar latents pass through convolutional waveform synthesis and stereo chunk assembly.","sources":["https://github.com/HeartMuLa/heartlib","https://huggingface.co/HeartMuLa/HeartCodec-oss-20260123"],"diagram":"heartcodec"},"ar-local-depth-decoder":{"id":"ar-local-depth-decoder","name":"Local codebook AR","role":"ar","summary":"A causal Transformer predicts seven residual codebooks after the semantic code. Learned depth positions and RMS-normalized attention distinguish it from the global Qwen3 model.","sources":["https://github.com/MiniMax-AI/MiniMax-Music3","https://huggingface.co/MiniMaxAI/MiniMax-Music3/tree/main/rvq_depth_decoder"],"diagram":"music3-depth"},"codec-flow-vae":{"id":"codec-flow-vae","name":"Flow-VAE waveform decoder","role":"codec","summary":"Continuous stereo latents enter a Snake convolutional decoder with transposed-convolution upsampling and dilated residual units. This stage does not perform RVQ lookup.","sources":["https://github.com/MiniMax-AI/MiniMax-Music3"],"diagram":"music3-vae"},"codec-midasheng-audio-tokenizer":{"id":"codec-midasheng-audio-tokenizer","name":"ConvNeXt / ISTFT audio decoder","role":"codec","summary":"Continuous latents are upsampled, refined by ConvNeXt blocks and projected to log-magnitude and phase for ISTFT waveform reconstruction. No discrete codebook lookup is used.","sources":["https://github.com/xiaomi-research/midashenglm-gen"],"diagram":"midasheng-codec"},"encoder-t5":{"id":"encoder-t5","name":"T5 text encoder","role":"encoder","summary":"SentencePiece text tokens feed a bidirectional T5 encoder with relative-position attention. Foundation uses the encoder features, not the T5 text decoder.","sources":["https://github.com/Stability-AI/stable-audio-tools"],"diagram":"stable-t5"},"codec-oobleck":{"id":"codec-oobleck","name":"Oobleck audio decoder","role":"codec","summary":"Continuous latents are decoded through transposed convolutions and dilated Snake residual units. No vector-quantized token lookup is required.","sources":["https://github.com/Stability-AI/stable-audio-tools"],"diagram":"oobleck-decoder"},"encoder-t5-t5gemma":{"id":"encoder-t5-t5gemma","name":"T5Gemma text encoder","role":"encoder","summary":"SA3 uses T5Gemma encoder features for text conditioning: bidirectional rotary attention, Gemma-style normalization and gated GELU feed-forward layers.","sources":["https://huggingface.co/stabilityai/stable-audio-3-small-sfx","https://huggingface.co/stabilityai/stable-audio-3-medium"],"diagram":"stable-t5gemma"},"codec-same":{"id":"codec-same","name":"SAME semantic-acoustic decoder","role":"codec","summary":"Latents are expanded with learned output tokens and local rotary Transformer blocks, then mapped back to waveform patches. SAME is not an Oobleck convolution-only decoder.","sources":["https://huggingface.co/stabilityai/stable-audio-3-small-sfx","https://github.com/Stability-AI/stable-audio-tools"],"diagram":"same-decoder"},"flow-conditional-flow":{"id":"flow-conditional-flow","name":"Multimodal flow DiT","role":"flow","summary":"Joint attention exchanges information between latent, visual, text and audio streams. Subsequent latent-only blocks predict flow under timestep, global and synchronization modulation.","sources":["https://github.com/xiaomi-research/controlfoley"],"diagram":"controlfoley-flow"},"codec-mel-latent-vae":{"id":"codec-mel-latent-vae","name":"Mel-latent convolutional VAE","role":"codec","summary":"Residual convolutions, middle attention and temporal upsampling reconstruct mel features. BigVGAN is separately required to turn these features into waveform audio.","sources":["https://github.com/xiaomi-research/controlfoley"],"diagram":"controlfoley-vae"},"flow-wan-s2v-dit":{"id":"flow-wan-s2v-dit","name":"Wan S2V flow DiT","role":"flow","summary":"Distilled video denoiser with 3D rotary self-attention, text cross-attention and audio injection. LiveAvatar advances video blocks while preserving prefix/history context.","sources":["https://github.com/Alibaba-Quark/LiveAvatar"],"diagram":"wan-s2v"},"codec-wan-video-vae":{"id":"codec-wan-video-vae","name":"Wan video VAE decoder","role":"codec","summary":"Causal 3D residual convolutions, spatial attention and spatiotemporal upsampling reconstruct RGB frames.","sources":["https://github.com/Wan-Video/Wan2.2"],"diagram":"wan-vae-decoder"},"encoder-qwen3-vl":{"id":"encoder-qwen3-vl","name":"Qwen3-VL text backbone","role":"encoder","summary":"H3 uses hidden states from the causal Qwen3-VL text backbone as conditioning. The current audio.cpp route does not run its vision tower.","sources":["https://huggingface.co/MiniMaxAI/MiniMax-H3"],"diagram":"h3-text"},"flow-multimodal-dit":{"id":"flow-multimodal-dit","name":"Joint audio/video flow DiT","role":"flow","summary":"Text, audio and video tokens share rotary attention while modality-specific time modulation conditions the flow solve.","sources":["https://huggingface.co/MiniMaxAI/MiniMax-H3"],"diagram":"h3-dit"},"encoder-mert2":{"id":"encoder-mert2","name":"MERT2 Conformer encoder","role":"encoder","summary":"ConvNeXt-style subsampling feeds rotary Conformer layers. A learned mixture of intermediate states supplies music representations to the score decoder.","sources":["https://huggingface.co/m-a-p/SheetSage2"],"diagram":"mert2"},"ar-sheetsage2-decoder":{"id":"ar-sheetsage2-decoder","name":"Cross-attention score AR decoder","role":"ar","summary":"Learned token/position embeddings feed a post-LayerNorm Transformer with causal self-attention, audio cross-attention and GELU feed-forward layers.","sources":["https://huggingface.co/m-a-p/SheetSage2"],"diagram":"sheetsage-decoder"},"head-abc-notation":{"id":"head-abc-notation","name":"Musical-event / ABC formatter","role":"dsp","summary":"Symbolic token IDs are decoded and assembled into timed musical events and readable ABC notation. This is postprocessing, not another learned network.","sources":["https://huggingface.co/m-a-p/SheetSage2"],"diagram":"sheetsage-events"},"encoder-audio-conditioning":{"id":"encoder-audio-conditioning","name":"Mel-prefix projection","role":"encoder","summary":"A linear layer projects mel frames to the decoder width. Dataset and instrument embeddings are appended as conditioning tokens; there is no separate deep audio encoder.","sources":["https://github.com/muscriptor/muscriptor"],"diagram":"muscriptor-prefix"},"ar-muscriptor-transformer":{"id":"ar-muscriptor-transformer","name":"Mel-prefix AR Transformer","role":"ar","summary":"A decoder-only pre-LayerNorm Transformer with sinusoidal positions and GELU feed-forward blocks generates musical event IDs from an audio prefix.","sources":["https://github.com/muscriptor/muscriptor"],"diagram":"muscriptor-ar"},"head-midi-events":{"id":"head-midi-events","name":"MIDI-event decoder","role":"dsp","summary":"Event tokens specify note timing, pitch and instrument. Chunk offsets and open-note state are resolved before writing MIDI or JSON; this is not a waveform codec.","sources":["https://github.com/muscriptor/muscriptor"],"diagram":"muscriptor-events"},"flow-diffllama":{"id":"flow-diffllama","name":"DiffLlama flow Transformer","role":"flow","summary":"Noncausal Llama-style rotary attention and SwiGLU layers use time-conditioned RMSNorm to predict mel velocity. This acoustic flow model is separate from Vevo2's Qwen2.5 AR token generator.","sources":["https://github.com/open-mmlab/Amphion/tree/main/models/svc/vevo2"],"diagram":"diffllama"},"encoder-xls-r":{"id":"encoder-xls-r","name":"XLS-R content encoder","role":"encoder","summary":"Multilingual Wav2Vec2-family waveform CNN and bidirectional Transformer. Seed-VC selects intermediate hidden features rather than decoding a transcript.","sources":["https://github.com/Plachtaa/seed-vc","https://huggingface.co/facebook/wav2vec2-xls-r-300m"],"diagram":"ssl-content"},"encoder-hybrid-transformer":{"id":"encoder-hybrid-transformer","name":"Hybrid Transformer","role":"encoder","summary":"Parallel waveform and spectral streams alternate self-attention and cross-domain attention. Neither branch is an AR text decoder.","sources":["https://github.com/facebookresearch/demucs/blob/main/demucs/transformer.py"],"diagram":"demucs-transformer"},"encoder-roformer":{"id":"encoder-roformer","name":"Axial RoFormer","role":"encoder","summary":"Alternating non-causal time-axis and band-axis Transformers with rotary positions and gated attention heads.","sources":["https://github.com/lucidrains/BS-RoFormer"],"diagram":"roformer-axial"},"codec-hifi-gan":{"id":"codec-hifi-gan","name":"HiFi-GAN vocoder","role":"codec","summary":"Mel features feed transposed-convolution upsampling and parallel dilated residual banks to synthesize waveform samples.","sources":["https://github.com/jik876/hifi-gan"],"diagram":"hifigan"},"encoder-rope-transformer":{"id":"encoder-rope-transformer","name":"RoPE Transformer","role":"encoder","summary":"Nemotron Diarization stacks mel frames, projects them and prepends speaker-cache context before a rotary Transformer encoder.","sources":["https://huggingface.co/nvidia/Nemotron-3-Diarization"],"diagram":"nemotron-diar-encoder"},"head-timestamp-prediction":{"id":"head-timestamp-prediction","name":"Timestamp classification","role":"head","summary":"Classifies timestamp slots, maps classes to time bins and repairs ordering to form word spans.","sources":["https://huggingface.co/Qwen/Qwen3-ForcedAligner-0.6B"],"diagram":"qwen-align-head"},"encoder-wav2vec2":{"id":"encoder-wav2vec2","name":"Wav2Vec2","role":"encoder","summary":"A strided waveform CNN, learned positional convolution and bidirectional Transformer produce acoustic frame representations. MMS uses the pre-norm variant.","sources":["https://huggingface.co/MahmoudAshraf/mms-300m-1130-forced-aligner/blob/main/config.json"],"diagram":"wav2vec2-mms"},"dsp-ctc-alignment":{"id":"dsp-ctc-alignment","name":"CTC forced alignment","role":"dsp","summary":"Dynamic programming finds a monotonic frame path through the known target sequence, then groups label spans into words.","sources":["https://huggingface.co/MahmoudAshraf/mms-300m-1130-forced-aligner/blob/main/config.json"],"diagram":"ctc-alignment"},"encoder-silero-cnn-lstm":{"id":"encoder-silero-cnn-lstm","name":"CNN + LSTM","role":"encoder","summary":"Silero's strided convolutional feature compression feeds a recurrent LSTM cell with state retained between windows.","sources":["https://github.com/snakers4/silero-vad"],"diagram":"silero-encoder"},"encoder-marblenet":{"id":"encoder-marblenet","name":"MarbleNet","role":"encoder","summary":"Residual temporal depthwise and pointwise convolutions, normalization and ReLU form a compact speech detector.","sources":["https://research.nvidia.com/publication/2020-10_marblenet-deep-1d-time-channel-separable-convolutional-neural-network-voice"],"diagram":"marblenet"},"encoder-depthwise-separable-cnn":{"id":"encoder-depthwise-separable-cnn","name":"Depthwise-separable CNN","role":"encoder","summary":"PulseVAD uses narrow temporal and pointwise convolutions, one projected residual block and a dilated final stage.","sources":["https://github.com/AydinAdnan/PulseVAD"],"diagram":"pulsevad"},"encoder-dual-path-zipformer":{"id":"encoder-dual-path-zipformer","name":"Dual-path Zipformer","role":"encoder","summary":"ZipEnhancer alternates frequency and time processing with down/up sampling and bypass paths.","sources":["https://zipenhancer.github.io/ZipEnhancer/"],"diagram":"zipenhancer"},"encoder-gtcrn":{"id":"encoder-gtcrn","name":"GTCRN","role":"encoder","summary":"Grouped temporal convolutions, channel shuffling and dual-path grouped GRUs predict a complex spectral mask.","sources":["https://github.com/Xiaobin-Rong/gtcrn"],"diagram":"gtcrn"},"encoder-umt5":{"id":"encoder-umt5","name":"UMT5 text encoder","role":"encoder","summary":"Multilingual SentencePiece input, relative-position attention and gated GELU feed-forward layers produce scene conditioning.","diagram":"umt5","sources":["https://github.com/Wan-Video/Wan2.2"]},"encoder-liveavatar-xlsr":{"id":"encoder-liveavatar-xlsr","name":"XLS-R speech encoder","role":"encoder","summary":"Wav2Vec2-family waveform convolutions and bidirectional Transformer features drive motion. Multiple hidden layers feed the audio adapter; no text is decoded.","diagram":"ssl-content","sources":["https://github.com/Alibaba-Quark/LiveAvatar"]},"encoder-wan-audio-condition":{"id":"encoder-wan-audio-condition","name":"Causal audio adapter","role":"encoder","summary":"Learned mixing of XLS-R layers and causal temporal convolutions form local attention tokens and global modulation features.","diagram":"wan-audio-condition"},"encoder-wan-vae":{"id":"encoder-wan-vae","name":"Wan video VAE encoder","role":"encoder","summary":"Causal 3D convolutions encode the reference image and video history into continuous spatiotemporal latents.","diagram":"wan-vae-encoder","sources":["https://github.com/Wan-Video/Wan2.2"]},"codec-h3-audio":{"id":"codec-h3-audio","name":"BigVGAN audio VAE decoder","role":"codec","summary":"Latent denormalization and projection feed a BigVGAN decoder separately for each audio channel.","diagram":"h3-audio","sources":["https://huggingface.co/MiniMaxAI/MiniMax-H3"]},"codec-h3-video":{"id":"codec-h3-video","name":"Transformer video VAE decoder","role":"codec","summary":"Latent patches and learned register tokens pass through rotary Transformer blocks before RGB patch reconstruction. Not the Wan convolutional VAE.","diagram":"h3-video","sources":["https://huggingface.co/MiniMaxAI/MiniMax-H3"]},"encoder-ace-qwen3":{"id":"encoder-ace-qwen3","name":"Qwen3 text conditioning","role":"encoder","summary":"A separate Qwen3 encodes the text prompt; its token embedding table also supplies lyric embeddings. Not the AR music planner.","diagram":"ace-text","sources":["https://github.com/ace-step/ACE-Step-1.5"]},"encoder-ace-condition":{"id":"encoder-ace-condition","name":"Lyric / timbre Transformers","role":"encoder","summary":"Projected text, rotary lyric encoding and reference-latent timbre encoding are packed into cross-attention context. XL uses a CLS timbre token and an independent encoder width.","diagram":"ace-condition","sources":["https://github.com/ace-step/ACE-Step-1.5"]},"encoder-ace-cover":{"id":"encoder-ace-cover","name":"Transformer + FSQ","role":"encoder","summary":"Attention pooling compresses VAE latent windows into discrete cover codes using finite scalar quantization.","diagram":"ace-cover"},"encoder-ace-detokenizer":{"id":"encoder-ace-detokenizer","name":"Transformer latent detokenizer","role":"encoder","summary":"FSQ codes are unpacked, projected and expanded through rotary Transformer blocks into latent-rate hints. This is not waveform decoding.","diagram":"ace-detokenizer"},"flow-ace-dit":{"id":"flow-ace-dit","name":"ACE-Step flow DiT","role":"flow","summary":"Time-modulated rotary self-attention, conditioning cross-attention and SwiGLU predict latent velocity. Standard and XL use checkpoint-specific dimensions and attention schedules.","diagram":"ace-dit","sources":["https://github.com/ace-step/ACE-Step-1.5"]},"encoder-ace-vae":{"id":"encoder-ace-vae","name":"Convolutional audio VAE encoder","role":"encoder","summary":"Periodic-activation residual convolution stages compress source waveforms to continuous audio latents.","diagram":"ace-vae-encoder"},"codec-ace-vae":{"id":"codec-ace-vae","name":"Convolutional audio VAE decoder","role":"codec","summary":"Snake residual units and transposed-convolution upsampling reconstruct stereo audio from continuous latents.","diagram":"ace-vae-decoder","sources":["https://github.com/ace-step/ACE-Step-1.5"]},"encoder-firered-understanding":{"id":"encoder-firered-understanding","name":"Whisper-style audio encoder","role":"encoder","summary":"Log-mel features feed convolutional subsampling and a noncausal Transformer, then a strided convolution/MLP adapter maps audio to language-model embeddings.","diagram":"firered-understanding","sources":["https://huggingface.co/FireRedTeam/FireRedAudio"]},"encoder-index-condition":{"id":"encoder-index-condition","name":"Conformer + Perceiver","role":"encoder","summary":"Relative-attention Conformer encodes semantic features; learned Perceiver queries compress them into fixed conditioning tokens. Emotion vectors can modify the emotion condition.","diagram":"index-condition","sources":["https://github.com/index-tts/index-tts"]},"encoder-index-emotion":{"id":"encoder-index-emotion","name":"Qwen3 emotion control","role":"encoder","summary":"Optional Qwen3 text generation maps an emotion description to control weights. It does not generate the speech-code sequence.","diagram":"index-emotion","sources":["https://github.com/index-tts/index-tts"]},"encoder-index-semantic-vq":{"id":"encoder-index-semantic-vq","name":"Semantic VQ","role":"encoder","summary":"Version 2 maps reference features through a ConvNeXt-style encoder to normalized codebook assignments. Generated code embeddings combine with projected AR hidden states before acoustic flow.","diagram":"index-vq","sources":["https://github.com/index-tts/index-tts"]},"encoder-index-enhanced-codec":{"id":"encoder-index-enhanced-codec","name":"ConvNeXt semantic decoder","role":"encoder","summary":"Version 2.5 decodes semantic codes through a ConvNeXt backbone and 2x temporal upsampling. This reconstructs semantic features, not a waveform.","diagram":"index-enhanced-codec","sources":["https://github.com/index-tts/index-tts"]},"frontend-sentencepiece":{"id":"frontend-sentencepiece","name":"SentencePiece tokenizer","role":"dsp","summary":"Model-specific text normalization, language markers and SentencePiece vocabulary map text to IDs. Vocabularies are not interchangeable.","diagram":"sentencepiece"},"encoder-w2v-bert":{"id":"encoder-w2v-bert","name":"Wav2Vec2-BERT 2.0","role":"encoder","summary":"Stacked filterbank features feed a relative-attention Conformer. This is not the waveform-CNN frontend of original Wav2Vec2 or HuBERT.","diagram":"w2v-bert","sources":["https://huggingface.co/facebook/w2v-bert-2.0"]},"encoder-confucius-ecapa":{"id":"encoder-confucius-ecapa","name":"ECAPA-TDNN speaker prompt","role":"encoder","summary":"SE-Res2Net TDNN and attentive statistics pooling turn Wav2Vec2-BERT features into the T2S speaker prompt. Separate from the flow model's CAM++ style encoder.","diagram":"confucius-speaker","sources":["https://github.com/netease-youdao/Confucius4-TTS/blob/main/confuciustts/llm/speaker_encoder.py"]},"encoder-glm-whisper-vq":{"id":"encoder-glm-whisper-vq","name":"Whisper-VQ speech tokenizer","role":"encoder","summary":"A Whisper-derived audio encoder is followed by temporal average pooling and codebook assignment. Its discrete speech tokens condition Llama and the acoustic flow.","diagram":"glm-whisper-vq","sources":["https://github.com/zai-org/GLM-TTS","https://huggingface.co/zai-org/GLM-TTS"]},"encoder-s3":{"id":"encoder-s3","name":"S3 speech tokenizer","role":"encoder","summary":"Log-mel subsampling and rotary attention with FSMN temporal memory produce finite-scalar-quantized speech tokens. Checkpoint dimensions and weights are specific to the model.","diagram":"s3-tokenizer","sources":["https://github.com/FunAudioLLM/CosyVoice","https://github.com/resemble-ai/chatterbox/tree/master/src/chatterbox/models/s3tokenizer"]},"flow-cosyvoice3":{"id":"flow-cosyvoice3","name":"Token-conditioned flow DiT","role":"flow","summary":"Speech-token embeddings are refined with lookahead convolutions and upsampled to mel rate. An adaptive-LayerNorm rotary DiT solves the conditional flow using prompt mel and speaker identity.","diagram":"cosyvoice3-flow","sources":["https://github.com/FunAudioLLM/CosyVoice/blob/main/cosyvoice/flow/DiT/dit.py","https://huggingface.co/FunAudioLLM/Fun-CosyVoice3-0.5B-2512"]},"encoder-redae":{"id":"encoder-redae","name":"RedAE audio encoder","role":"encoder","summary":"Waveform patches pass through causal Qwen3-style blocks, followed by local CLS-token downsampling and a continuous bottleneck projection. This is not discrete speech tokenization.","diagram":"redae-encoder","sources":["https://github.com/FireRedTeam/FireRedTTS3/blob/main/fireredtts3/redae/redae.py"]},"encoder-firered-patch":{"id":"encoder-firered-patch","name":"Latent-patch Transformer","role":"encoder","summary":"A small noncausal rotary Transformer summarizes each group of RedAE latent frames with a CLS token and projects it into the Qwen3 prompt space. Generated patches use the same feedback path.","diagram":"firered-patch","sources":["https://github.com/FireRedTeam/FireRedTTS3/blob/main/fireredtts3/llm/patch_encoder.py"]},"frontend-vieneu-phones":{"id":"frontend-vieneu-phones","name":"SEA-G2P / phoneme tokens","role":"dsp","summary":"Optional SEA-G2P converts text to phonemes when its dictionary is configured. Otherwise callers supply phonemes directly; the model vocabulary encodes the resulting phone sequence.","diagram":"vieneu-frontend","sources":["https://github.com/pnnbao97/VieNeu-TTS","https://huggingface.co/pnnbao-ump/VieNeu-TTS-v3-Turbo"]},"encoder-vieneu-anchor":{"id":"encoder-vieneu-anchor","name":"Speaker embedding projection","role":"encoder","summary":"A supplied 192-dimensional speaker vector is projected and normalized into an additive prompt anchor. This is not an in-process CAM++ audio encoder.","diagram":"vieneu-anchor","sources":["https://github.com/pnnbao97/VieNeu-TTS"]},"ar-audio8-slow":{"id":"ar-audio8-slow","name":"Qwen-style temporal AR","role":"ar","summary":"Audio8's 0.6B rotary GQA decoder uses RMSNorm and SwiGLU, with biased QKV projections and no Q/K normalization. It predicts semantic audio tokens along time.","diagram":"audio8-slow","sources":["https://huggingface.co/Edge0/Audio8-TTS-Preview-0.6b/raw/main/config.json"]},"ar-audio8-fast":{"id":"ar-audio8-fast","name":"Rotary depth AR","role":"ar","summary":"A short causal decoder expands the semantic token into the residual codebooks for one frame. Its state is reset at each frame boundary.","diagram":"audio8-fast","sources":["https://huggingface.co/Edge0/Audio8-TTS-Preview-0.6b/raw/main/config.json"]},"ar-audio8-falcon":{"id":"ar-audio8-falcon","name":"Falcon-H1 hybrid AR","role":"ar","summary":"The alternate slow-backbone implementation combines parallel Mamba2 and grouped-query attention branches, followed by a gated feed-forward sublayer. It is not the packaged 0.6B route.","diagram":"audio8-falcon","sources":["https://github.com/Audio8-AI/Audio8_TTS","https://huggingface.co/tiiuae/Falcon-H1-0.5B-Base"]},"encoder-audio8-codec":{"id":"encoder-audio8-codec","name":"Audio8 reference codec encoder","role":"encoder","summary":"Snake CNN downsampling, windowed attention and ConvNeXt stages produce semantic and residual reference codes for the AR prompt.","diagram":"fish-dac-encoder","sources":["https://github.com/Audio8-AI/Audio8_TTS"]},"ar-fish-slow":{"id":"ar-fish-slow","name":"Fish temporal AR Transformer","role":"ar","summary":"The slow causal decoder consumes text and summed audio-codebook embeddings. It predicts semantic codes and preserves the hidden state used by the fast decoder.","diagram":"fish-slow","sources":["https://huggingface.co/fishaudio/s2-pro"]},"ar-fish-fast":{"id":"ar-fish-fast","name":"Fish depth AR Transformer","role":"ar","summary":"The fast decoder resets per audio frame and generates the remaining residual codebooks from the slow state and semantic code.","diagram":"fish-fast","sources":["https://huggingface.co/fishaudio/s2-pro"]},"ar-moss-nano-depth":{"id":"ar-moss-nano-depth","name":"Nano depth AR Transformer","role":"ar","summary":"A local rotary Transformer predicts the text/control token and then audio codebooks for each frame, conditioned on the global state.","diagram":"moss-nano-depth","sources":["https://huggingface.co/OpenMOSS-Team/MOSS-TTS-Nano-100M"]},"encoder-moss-reference":{"id":"encoder-moss-reference","name":"MOSS causal audio encoder","role":"encoder","summary":"Waveform patches pass through multirate causal Transformers, then residual quantization produces reference codebooks. No separate speaker embedding network is required by this conditioning route.","diagram":"moss-codec-encoder","sources":["https://huggingface.co/OpenMOSS-Team/MOSS-Audio-Tokenizer"]},"encoder-music3-hidden-fusion":{"id":"encoder-music3-hidden-fusion","name":"Global / local hidden fusion","role":"encoder","summary":"Learned normalized weights mix the global and local frame states. A temporal convolution and nearest-neighbor resampling align them with the flow latent rate.","diagram":"music3-fusion","sources":["https://github.com/MiniMax-AI/MiniMax-Music3"]},"flow-music3":{"id":"flow-music3","name":"Latent flow Transformer","role":"flow","summary":"Hidden-state conditioning is concatenated with noisy audio latents. A prepended time token conditions noncausal rotary attention and SwiGLU blocks.","diagram":"music3-flow","sources":["https://huggingface.co/docs/diffusers/main/en/api/pipelines/minimax_music3"]},"ar-midasheng-qwen3":{"id":"ar-midasheng-qwen3","name":"Qwen3 continuous-patch AR","role":"ar","summary":"Qwen3-1.7B consumes text embeddings and projected generated patches. Hidden states condition flow generation and a two-logit stop head, not a discrete audio vocabulary.","diagram":"midasheng-ar","sources":["https://huggingface.co/mispeech/midashenglm-gen","https://github.com/xiaomi-research/midashenglm-gen"]},"flow-midasheng-patch":{"id":"flow-midasheng-patch","name":"Patch flow Transformer","role":"flow","summary":"Time and AR hidden conditioning form a token alongside the previous patch and current noisy patch. A rotary Transformer predicts the current patch velocity.","diagram":"midasheng-flow","sources":["https://github.com/xiaomi-research/midashenglm-gen"]},"encoder-irodori-speaker":{"id":"encoder-irodori-speaker","name":"Reference-latent Transformer","role":"encoder","summary":"Continuous DAC-VAE reference latents are patched, projected and encoded with gated rotary Transformer blocks into speaker conditioning states.","diagram":"irodori-speaker","sources":["https://github.com/Aratako/Irodori-TTS"]},"head-irodori-duration":{"id":"head-irodori-duration","name":"Conditional duration predictor","role":"head","summary":"Text states, speaker conditioning and optional caption conditioning predict output length. An explicit duration can override this estimate.","diagram":"irodori-duration","sources":["https://huggingface.co/Aratako/Irodori-TTS-v4.1-Small"]},"encoder-chatterbox-voice":{"id":"encoder-chatterbox-voice","name":"LSTM speaker encoder","role":"encoder","summary":"Overlapping mel windows enter a three-layer LSTM. Projected final states are normalized and averaged into the speaker vector used by T3.","diagram":"chatterbox-voice","sources":["https://github.com/resemble-ai/chatterbox/blob/master/src/chatterbox/models/voice_encoder/voice_encoder.py"]},"encoder-chatterbox-s3":{"id":"encoder-chatterbox-s3","name":"S3 speech tokenizer","role":"encoder","summary":"A rotary attention encoder with FSMN temporal memory converts log-mel features into finite-scalar-quantized speech tokens.","diagram":"chatterbox-s3","sources":["https://github.com/resemble-ai/chatterbox/tree/master/src/chatterbox/models/s3tokenizer"]},"encoder-chatterbox-prompt":{"id":"encoder-chatterbox-prompt","name":"Perceiver prompt resampler","role":"encoder","summary":"Learned queries compress reference speech-token embeddings into a fixed-length conditioning sequence for regular Chatterbox's T3 decoder.","diagram":"chatterbox-perceiver","sources":["https://github.com/resemble-ai/chatterbox/tree/master/src/chatterbox/models/t3"]},"encoder-auk-audio":{"id":"encoder-auk-audio","name":"Qwen2.5-Omni audio encoder","role":"encoder","summary":"Log-mel features pass through convolutional subsampling and windowed Transformer encoding, followed by pooling and projection into thinker embeddings.","diagram":"auk-audio","sources":["https://github.com/Tencent-Hunyuan/AuK"]},"encoder-auk-vae":{"id":"encoder-auk-vae","name":"Residual CNN VAE encoder","role":"encoder","summary":"Source or reference waveforms are encoded into continuous latent frames. These form a separate audio prefix for the flow model.","diagram":"auk-vae-encoder","sources":["https://github.com/Tencent-Hunyuan/AuK/blob/main/src/auk/model/vae/bigvgan_flow_vae.py"]},"ar-breeze-depth":{"id":"ar-breeze-depth","name":"Llama-style depth AR","role":"ar","summary":"A temporal hidden state and first audio code seed a smaller causal decoder that predicts the remaining codebooks within each audio frame.","diagram":"breeze-depth","sources":["https://huggingface.co/BreezeBlue/Breeze-TTS-2/raw/main/config.json"]},"encoder-breeze-codec":{"id":"encoder-breeze-codec","name":"Mimi-derived speech encoder","role":"encoder","summary":"A causal waveform CNN and rotary Transformer feed semantic and acoustic quantizers. Reference codes provide voice conditioning; the reference transcript accompanies them.","diagram":"breeze-reference","sources":["https://huggingface.co/BreezeBlue/Breeze-TTS-2"]},"encoder-stable-timing":{"id":"encoder-stable-timing","name":"Fourier timing conditioner","role":"encoder","summary":"Normalized duration, and start time for Foundation, pass through Fourier features and learned projections. These embeddings condition both cross-attention and the global diffusion state.","diagram":"stable-timing","sources":["https://github.com/Stability-AI/stable-audio-tools"]},"flow-stable-foundation":{"id":"flow-stable-foundation","name":"Foundation rectified-flow DiT","role":"flow","summary":"A prepended time/duration token and text cross-attention condition a noncausal rotary Transformer with LayerNorm and SwiGLU. Audio initialization changes the starting noisy latent.","diagram":"stable-foundation-flow","sources":["https://github.com/Stability-AI/stable-audio-tools"]},"flow-stable3-rf-dit":{"id":"flow-stable3-rf-dit","name":"Stable Audio 3 rectified-flow DiT","role":"flow","summary":"Memory tokens, adaptive RMS normalization, text cross-attention and optional masked-audio conditioning drive latent diffusion. Attention can be differential according to the checkpoint.","diagram":"stable3-flow","sources":["https://github.com/Stability-AI/stable-audio-tools","https://huggingface.co/stabilityai/stable-audio-3-medium"]},"encoder-oobleck":{"id":"encoder-oobleck","name":"Oobleck audio encoder","role":"encoder","summary":"Strided residual convolutions with periodic Snake activations compress source waveforms into a continuous variational latent for initialization.","diagram":"oobleck-encoder","sources":["https://github.com/Stability-AI/stable-audio-tools"]},"encoder-same":{"id":"encoder-same","name":"SAME semantic-acoustic encoder","role":"encoder","summary":"Waveform patches and learned summary tokens pass through local rotary Transformer blocks. Summary outputs are projected and scaled into the diffusion latent space.","diagram":"same-encoder","sources":["https://huggingface.co/stabilityai/stable-audio-3-small-sfx","https://github.com/Stability-AI/stable-audio-tools"]},"encoder-mira-ecapa-perceiver":{"id":"encoder-mira-ecapa-perceiver","name":"ECAPA / Perceiver speaker tokenizer","role":"encoder","summary":"Reference mel features pass through SE-Res2 TDNN blocks and a learned-query Perceiver. Finite scalar quantization produces 32 context tokens used by both the language model and acoustic processor.","diagram":"mira-speaker","sources":["https://github.com/ysharma3501/MiraTTS"]},"encoder-dots-patch":{"id":"encoder-dots-patch","name":"Causal semantic patch Transformer","role":"encoder","summary":"Downsamples continuous audio latents and compresses each patch into a Qwen input embedding. It processes reference/source patches and each generated patch; it is not a VQ tokenizer.","diagram":"dots-patch","sources":["https://github.com/studio-dots-ai/dots.tts"]},"encoder-dots-vae":{"id":"encoder-dots-vae","name":"Dots continuous AudioVAE encoder","role":"encoder","summary":"Causal residual convolutions and an LSTM bottleneck produce a latent distribution from reference/source speech. Sampling and normalization supply continuous acoustic patches.","diagram":"dots-vae-encoder","sources":["https://github.com/studio-dots-ai/dots.tts"]},"frontend-heartmula-text":{"id":"frontend-heartmula-text","name":"Lyrics / tags tokenizer","role":"dsp","summary":"Tokenized tags and lyrics form the text prompt. A projected zero continuous-conditioning vector occupies the reserved MuQ position; this path does not run MuQ or HeartCLAP.","diagram":"heartmula-prompt","sources":["https://github.com/HeartMuLa/heartlib"]},"ar-heartmula-temporal":{"id":"ar-heartmula-temporal","name":"Llama-style temporal AR Transformer","role":"ar","summary":"A Llama 3.2-style causal backbone predicts the first audio codebook and a hidden state per frame. Text and audio-code embeddings share the temporal sequence; architecture attribution does not imply pretrained Llama text weights.","diagram":"heartmula-temporal","sources":["https://github.com/HeartMuLa/heartlib/blob/main/src/heartlib/heartmula/modeling_heartmula.py"]},"ar-heartmula-depth":{"id":"ar-heartmula-depth","name":"Llama-style codebook AR Transformer","role":"ar","summary":"A smaller causal Transformer predicts remaining codebooks from the temporal hidden state and first code. Its cache resets for each audio frame, unlike the temporal backbone.","diagram":"heartmula-depth","sources":["https://github.com/HeartMuLa/heartlib/blob/main/src/heartlib/heartmula/modeling_heartmula.py"]},"frontend-personaplex-text":{"id":"frontend-personaplex-text","name":"SentencePiece persona tokenizer","role":"dsp","summary":"The system/persona prompt is wrapped in the expected control tags and tokenized as a conversation prefix. It is an instruction, not a literal TTS transcript.","diagram":"personaplex-text","sources":["https://github.com/NVIDIA/personaplex"]},"encoder-personaplex-prompt":{"id":"encoder-personaplex-prompt","name":"Stored voice-prompt embeddings","role":"encoder","summary":"A packaged voice ID selects prompt embeddings and audio-delay state, replayed into the temporal model. This path does not encode a reference waveform.","diagram":"personaplex-preset","sources":["https://github.com/NVIDIA/personaplex"]},"encoder-mimi":{"id":"encoder-mimi","name":"Mimi speech encoder","role":"encoder","summary":"Causal SEANet-style convolutions, a rotary Transformer and rate reduction feed separate semantic and acoustic quantization branches.","diagram":"mimi-encoder","sources":["https://github.com/kyutai-labs/moshi"]},"ar-moshi-temporal":{"id":"ar-moshi-temporal","name":"Moshi temporal AR Transformer","role":"ar","summary":"Summed text and delayed audio-stream embeddings drive a causal RMSNorm/RoPE/SwiGLU Transformer. It predicts a text token and a state for within-frame audio generation.","diagram":"moshi-temporal","sources":["https://github.com/NVIDIA/personaplex"]},"encoder-openclip-text":{"id":"encoder-openclip-text","name":"OpenCLIP text Transformer","role":"encoder","summary":"Byte-level BPE and learned positions feed a causal text Transformer. Token features condition generation; this branch does not generate text.","diagram":"controlfoley-clip-text","sources":["https://github.com/xiaomi-research/controlfoley"]},"encoder-openclip-vision":{"id":"encoder-openclip-vision","name":"OpenCLIP vision Transformer","role":"encoder","summary":"Sampled video frames pass through patch embedding and a vision Transformer. Projected class-token features provide visual semantics.","diagram":"controlfoley-clip-vision","sources":["https://github.com/xiaomi-research/controlfoley"]},"encoder-cavmae-visual":{"id":"encoder-cavmae-visual","name":"CAV-MAE-ST visual Transformer","role":"encoder","summary":"Patch, position and modality embeddings feed visual and shared Transformer blocks from an audio-visual pretrained encoder.","diagram":"controlfoley-cavmae","sources":["https://github.com/xiaomi-research/controlfoley"]},"encoder-synchformer":{"id":"encoder-synchformer","name":"Synchformer timing encoder","role":"encoder","summary":"Video tubelets pass through factorized temporal/spatial attention and spatial aggregation, providing time-aligned conditioning for sound generation.","diagram":"controlfoley-synchformer","sources":["https://github.com/xiaomi-research/controlfoley"]},"encoder-clap-audio":{"id":"encoder-clap-audio","name":"CLAP audio Swin Transformer","role":"encoder","summary":"A log-mel frontend and hierarchical shifted-window Transformer produce a projected audio-semantic embedding from the reference sound.","diagram":"controlfoley-clap","sources":["https://github.com/xiaomi-research/controlfoley"]},"encoder-musicgen-style":{"id":"encoder-musicgen-style","name":"MERT / MusicGen style encoder","role":"encoder","summary":"Raw-audio MERT features feed a style Transformer, codebook quantization and temporal reduction. ControlFoley averages the projected style tokens for global timbre conditioning.","diagram":"controlfoley-style","sources":["https://github.com/xiaomi-research/controlfoley"]},"frontend-sheetsage-mel":{"id":"frontend-sheetsage-mel","name":"Normalized music mel frontend","role":"dsp","summary":"Resampling and channel mixing precede windowed STFT and a checkpoint mel filterbank. Per-bin training mean and standard deviation normalize log-power features.","diagram":"sheetsage-mel","sources":["https://huggingface.co/m-a-p/SheetSage2"]},"encoder-dramabox-connector":{"id":"encoder-dramabox-connector","name":"Gated-attention prompt connector","role":"encoder","summary":"Learned registers fill unused prompt positions. A noncausal rotary Transformer with Q/K normalization, head gates and GELU feed-forward layers prepares text memory for the audio DiT.","diagram":"dramabox-connector","sources":["https://github.com/resemble-ai/DramaBox"]},"flow-dramabox-dit":{"id":"flow-dramabox-dit","name":"LTX audio flow DiT","role":"flow","summary":"Timestep-modulated self-attention, text cross-attention and feed-forward blocks predict audio-latent flow. Optional reference latents condition generation without a reference transcript.","diagram":"dramabox-dit","sources":["https://huggingface.co/ResembleAI/Dramabox"]},"encoder-dramabox-vae":{"id":"encoder-dramabox-vae","name":"Causal mel AudioVAE encoder","role":"encoder","summary":"PixelNorm residual 2D convolutions downsample reference mel features. The posterior mean provides conditioning latents; these are not discrete codec IDs.","diagram":"dramabox-vae-encoder","sources":["https://github.com/resemble-ai/DramaBox"]},"codec-dramabox-vae":{"id":"codec-dramabox-vae","name":"Causal mel AudioVAE decoder","role":"codec","summary":"A causal convolutional residual decoder upsamples audio latents into stereo mel features. Waveform synthesis is a separate BigVGAN stage.","diagram":"dramabox-vae-decoder","sources":["https://github.com/resemble-ai/DramaBox"]},"codec-dramabox-bwe":{"id":"codec-dramabox-bwe","name":"BigVGAN bandwidth extension","role":"codec","summary":"A second BigVGAN predicts a high-rate residual from the low-rate waveform's mel features. The output adds a resampled waveform skip and clamps the sum.","diagram":"dramabox-bwe","sources":["https://github.com/resemble-ai/DramaBox"]},"frontend-magpie-text":{"id":"frontend-magpie-text","name":"Language-specific text frontend","role":"dsp","summary":"Checkpoint-specific IPA/pinyin phoneme, character or byte tokenization. A ByT5-style byte tokenizer does not imply a pretrained T5 text encoder.","diagram":"magpie-text","sources":["https://huggingface.co/nvidia/magpie_tts_multilingual_357m"]},"encoder-magpie-text":{"id":"encoder-magpie-text","name":"Causal text Transformer","role":"encoder","summary":"Learned token and position embeddings feed a causal Transformer with convolutional feed-forward layers. Its outputs supply cross-attention memory.","diagram":"magpie-encoder","sources":["https://huggingface.co/nvidia/magpie_tts_multilingual_357m"]},"ar-magpie-temporal":{"id":"ar-magpie-temporal","name":"Cross-attention temporal AR","role":"ar","summary":"Preset speaker context and prior audio-frame embeddings feed causal attention. Cross-attention reads encoded text, with alignment priors guiding generation.","diagram":"magpie-temporal","sources":["https://huggingface.co/nvidia/magpie_tts_multilingual_357m"]},"ar-magpie-local":{"id":"ar-magpie-local","name":"Local codebook AR Transformer","role":"ar","summary":"A separate local causal Transformer predicts codebooks within each frame stack from the temporal decoder state. Each codebook position has its own output projection.","diagram":"magpie-local","sources":["https://huggingface.co/nvidia/magpie_tts_multilingual_357m"]},"encoder-magpie-preset":{"id":"encoder-magpie-preset","name":"Stored speaker context","role":"encoder","summary":"A speaker ID selects a baked context embedding. The current integration does not run a reference-audio encoder for voice cloning.","diagram":"magpie-preset","sources":["https://huggingface.co/nvidia/magpie_tts_multilingual_357m"]},"encoder-sopro-semantic":{"id":"encoder-sopro-semantic","name":"Whisper-style semantic encoder","role":"encoder","summary":"Log-mel features pass through convolutional subsampling and a pre-LayerNorm Transformer. A scalar-level classification head packs semantic token IDs for the AR prompt.","diagram":"sopro-semantic","sources":["https://github.com/samuel-vitorino/sopro"]},"encoder-sopro-speaker":{"id":"encoder-sopro-speaker","name":"Gated CNN speaker/style encoder","role":"encoder","summary":"A multiscale depthwise residual CNN uses squeeze-excitation and pooled statistics to produce speaker identity and style conditioning.","diagram":"sopro-speaker","sources":["https://github.com/samuel-vitorino/sopro"]},"frontend-sopro-text":{"id":"frontend-sopro-text","name":"SentencePiece text tokenizer","role":"dsp","summary":"Minimal text normalization and language marking precede SentencePiece tokenization. No phonemization is required.","diagram":"sopro-text","sources":["https://github.com/samuel-vitorino/sopro/blob/main/src/sopro/text.py"]},"encoder-echo-text":{"id":"encoder-echo-text","name":"Byte-level text Transformer","role":"encoder","summary":"Normalized UTF-8 bytes and a BOS token feed a learned embedding and bidirectional gated-attention Transformer. No pretrained language model or phonemizer is involved.","diagram":"echo-text","sources":["https://github.com/jordandare/echo-tts/blob/main/model.py"]},"encoder-echo-reference":{"id":"encoder-echo-reference","name":"Reference-latent Transformer","role":"encoder","summary":"PCA-compressed Fish DAC reference latents are patchified and encoded by a causal Transformer to provide speaker conditioning keys and values.","diagram":"echo-reference","sources":["https://github.com/jordandare/echo-tts/blob/main/model.py"]},"encoder-fish-dac":{"id":"encoder-fish-dac","name":"Fish DAC audio encoder","role":"encoder","summary":"Snake residual downsampling, windowed attention and ConvNeXt rate conversion feed a residual vector quantizer. Fish uses reference code IDs; Echo consumes reconstructed quantized latents with its own codec weights.","diagram":"fish-dac-encoder","sources":["https://github.com/fishaudio/fish-speech","https://github.com/jordandare/echo-tts"]},"codec-echo-fish-dac":{"id":"codec-echo-fish-dac","name":"Fish DAC latent decoder","role":"codec","summary":"Echo's continuous output is mapped back from PCA space into the Fish DAC latent representation. The codec runs its Transformer/rate-conversion and waveform decoder without looking up generated code IDs.","diagram":"fish-dac-latent-decoder","sources":["https://github.com/jordandare/echo-tts"]},"flow-echo-dit":{"id":"flow-echo-dit","name":"Echo joint-attention DiT","role":"flow","summary":"Low-rank adaptive normalization conditions a diffusion Transformer on time. Joint attention attends to noisy latents, encoded text and reference speaker states, with separate text and speaker guidance.","diagram":"echo-dit","sources":["https://github.com/jordandare/echo-tts/blob/main/model.py"]},"dsp-echo-pca":{"id":"dsp-echo-pca","name":"PCA latent transform","role":"dsp","summary":"Centers, projects and scales codec latents for diffusion; the inverse transform reconstructs the codec representation before waveform decoding.","diagram":"echo-pca","sources":["https://github.com/jordandare/echo-tts"]},"encoder-higgs-v2-reference":{"id":"encoder-higgs-v2-reference","name":"Higgs Audio V2 reference encoder","role":"encoder","summary":"HuBERT semantic features and a Snake residual acoustic CNN are aligned, concatenated and projected before residual vector quantization.","diagram":"higgs-v2-reference","sources":["https://github.com/k2-fsa/OmniVoice/blob/main/omnivoice/models/omnivoice.py"]},"head-omnivoice-duration":{"id":"head-omnivoice-duration","name":"Rule-based duration estimate","role":"dsp","summary":"Text weighting, optional reference transcript/audio duration and speed determine the target audio-code length. This is not a learned duration predictor.","diagram":"omnivoice-duration","sources":["https://github.com/k2-fsa/OmniVoice/blob/main/omnivoice/utils/duration.py"]},"encoder-voxcpm-local":{"id":"encoder-voxcpm-local","name":"MiniCPM local patch encoder","role":"encoder","summary":"Projects an AudioVAE patch, prepends a learned summary token and applies noncausal MiniCPM Transformer blocks. The summary token is projected into the temporal language model.","diagram":"voxcpm-local","sources":["https://github.com/OpenBMB/VoxCPM"]},"ar-voxcpm-hierarchy":{"id":"ar-voxcpm-hierarchy","name":"MiniCPM hierarchical AR","role":"ar","summary":"A text-semantic causal LM and a residual acoustic causal LM condition each generated audio patch. Scalar quantization constrains the semantic hidden state; these are not waveform codec token IDs.","diagram":"voxcpm-ar","sources":["https://huggingface.co/openbmb/VoxCPM-0.5B","https://huggingface.co/openbmb/VoxCPM2"]},"flow-voxcpm1-local":{"id":"flow-voxcpm1-local","name":"VoxCPM1 local diffusion Transformer","role":"flow","summary":"Summed semantic/acoustic conditioning is added to the time token. Noncausal MiniCPM blocks predict patch velocity from this token, the previous patch and the noisy patch.","diagram":"voxcpm1-flow","sources":["https://github.com/OpenBMB/VoxCPM"]},"flow-voxcpm2-local":{"id":"flow-voxcpm2-local","name":"VoxCPM2 local diffusion Transformer","role":"flow","summary":"Semantic and residual acoustic conditioning are separate prefix tokens alongside time, previous-patch and noisy-patch embeddings. Noncausal MiniCPM blocks predict flow velocity.","diagram":"voxcpm2-flow","sources":["https://huggingface.co/openbmb/VoxCPM2"]},"encoder-voxcpm-audiovae":{"id":"encoder-voxcpm-audiovae","name":"AudioVAE reference encoder","role":"encoder","summary":"Causal residual convolutions with Snake activations downsample reference audio to continuous latent patches. Inference uses the latent mean, without a discrete codebook.","diagram":"voxcpm-vae-encoder","sources":["https://github.com/OpenBMB/VoxCPM"]},"frontend-pocket-text":{"id":"frontend-pocket-text","name":"SentencePiece text embedding","role":"dsp","summary":"Text normalization and SentencePiece IDs feed a learned embedding table for the FlowLM prefix.","diagram":"pocket-text","sources":["https://github.com/kyutai-labs/pocket-tts"]},"encoder-pocket-mimi":{"id":"encoder-pocket-mimi","name":"Mimi reference encoder","role":"encoder","summary":"Causal convolutional and Transformer encoding produces continuous reference latents, projected into FlowLM prompt embeddings. It does not quantize them to token IDs.","diagram":"pocket-reference","sources":["https://github.com/kyutai-labs/pocket-tts"]},"codec-pocket-mimi":{"id":"codec-pocket-mimi","name":"Mimi continuous decoder","role":"codec","summary":"Continuous audio latents feed a projection, rate upsampling, causal rotary Transformer and SEANet-style convolutional waveform decoder. No codebook lookup is required.","diagram":"pocket-mimi-decoder","sources":["https://github.com/kyutai-labs/pocket-tts"]},"encoder-neutts-preset":{"id":"encoder-neutts-preset","name":"Stored speech-code prompt","role":"encoder","summary":"Select a packaged speaker's transcript and precomputed NeuCodec codes. No reference-audio encoder runs in this route.","diagram":"neutts-preset"},"encoder-dac-reference":{"id":"encoder-dac-reference","name":"DAC audio encoder","role":"encoder","summary":"Snake residual convolutional downsampling followed by projected residual vector quantization produces reference audio codes.","diagram":"dac-encoder","sources":["https://github.com/descriptinc/descript-audio-codec"]},"encoder-oute-alignment":{"id":"encoder-oute-alignment","name":"Qwen3 forced alignment","role":"encoder","summary":"In audio.cpp's OuteTTS cloning path, a Qwen3 Forced Aligner maps the reference transcript to audio spans before prompt construction. This is a companion model, not an OuteTTS backbone layer.","diagram":"oute-aligner","sources":["https://huggingface.co/Qwen/Qwen3-ForcedAligner-0.6B"]},"frontend-oute-profile":{"id":"frontend-oute-profile","name":"Aligned voice prompt","role":"dsp","summary":"Word spans, DAC codes and reference acoustic statistics are serialized into the OuteTTS token prompt.","diagram":"oute-profile","sources":["https://huggingface.co/OuteAI/Llama-OuteTTS-1.0-1B"]},"frontend-f5-text":{"id":"frontend-f5-text","name":"Character / dialect tokenizer","role":"dsp","summary":"Checkpoint vocabulary maps normalized text to IDs. Packaged Habibi checkpoints add dialect tokens; this is not a pretrained language-model tokenizer.","diagram":"f5-tokenizer","sources":["https://github.com/SWivid/F5-TTS","https://huggingface.co/SWivid/Habibi-TTS"]},"flow-f5-dit":{"id":"flow-f5-dit","name":"DiT mel flow","role":"flow","summary":"A rotary Diffusion Transformer jointly conditions on text features, reference mel and noisy mel. Time-adaptive LayerNorm and gated residuals drive continuous mel generation.","diagram":"f5-dit","sources":["https://github.com/SWivid/F5-TTS"]},"frontend-zipvoice-text":{"id":"frontend-zipvoice-text","name":"Emilia phoneme tokenizer","role":"dsp","summary":"The Emilia frontend combines Chinese segmentation/pinyin with English eSpeak phonemes and checkpoint vocabulary IDs. Alternate tokenizer modes are package-dependent.","diagram":"zipvoice-tokenizer","sources":["https://github.com/k2-fsa/ZipVoice"]},"encoder-zipvoice-text":{"id":"encoder-zipvoice-text","name":"Zipformer text encoder","role":"encoder","summary":"Noncausal Zipformer layers encode prompt and target text tokens. Features are expanded to mel-frame duration estimated from the reference.","diagram":"zipvoice-text","sources":["https://github.com/k2-fsa/ZipVoice"]},"frontend-audiosr-bandlimit":{"id":"frontend-audiosr-bandlimit","name":"Bandwidth analysis + log-mel","role":"dsp","summary":"Estimate the source cutoff, prepare a low-pass reference and extract mel conditioning.","diagram":"audiosr-frontend","sources":["https://github.com/haoheliu/versatile_audio_super_resolution"]},"encoder-audiosr-vae":{"id":"encoder-audiosr-vae","name":"Mel VAE encoder","role":"encoder","summary":"A residual 2D convolutional encoder with bottleneck attention maps the low-bandwidth mel spectrogram to a sampled Gaussian condition latent.","diagram":"audiosr-vae-encoder","sources":["https://github.com/haoheliu/versatile_audio_super_resolution"]},"flow-audiosr-unet":{"id":"flow-audiosr-unet","name":"Latent diffusion U-Net","role":"flow","summary":"A timestep-conditioned residual U-Net with spatial Transformers denoises mel latents. Low-bandwidth latents enter by channel concatenation, not text cross-attention.","diagram":"audiosr-unet","sources":["https://github.com/haoheliu/versatile_audio_super_resolution"]},"codec-audiosr-vae":{"id":"codec-audiosr-vae","name":"Mel VAE decoder","role":"codec","summary":"Bottleneck attention and residual convolutional upsampling reconstruct a high-bandwidth mel spectrogram; a separate HiFi-GAN generates audio.","diagram":"audiosr-vae-decoder","sources":["https://github.com/haoheliu/versatile_audio_super_resolution"]},"dsp-audiosr-preserve":{"id":"dsp-audiosr-preserve","name":"Low-band reconstruction","role":"dsp","summary":"Preserve low-frequency reference information in mel space and again in waveform STFT space after neural synthesis.","diagram":"audiosr-preserve","sources":["https://github.com/haoheliu/versatile_audio_super_resolution"]},"encoder-campplus":{"id":"encoder-campplus","name":"CAMPPlus speaker encoder","role":"encoder","summary":"Residual 2D convolutions and densely connected context-aware TDNN blocks summarize reference filterbanks into a speaker embedding.","diagram":"campplus","sources":["https://github.com/alibaba-damo-academy/3D-Speaker"]},"encoder-seed-hubert-astral":{"id":"encoder-seed-hubert-astral","name":"HuBERT + ASTRAL BSQ","role":"encoder","summary":"HuBERT-Large speech features pass through an ASTRAL ConvNeXt encoder and binary spherical quantization. Seed-VC v2's current audio.cpp conversion route uses the wide codebook directly, without AR resynthesis.","diagram":"seed-astral","sources":["https://github.com/Plachtaa/seed-vc"]},"encoder-seed-length":{"id":"encoder-seed-length","name":"Convolutional length regulator","role":"encoder","summary":"Content features or embedded tokens are resized to acoustic-frame length, optionally combined with pitch embeddings, and processed by normalized convolutions.","diagram":"seed-length","sources":["https://github.com/Plachtaa/seed-vc"]},"flow-seed-uvit":{"id":"flow-seed-uvit","name":"U-ViT acoustic flow","role":"flow","summary":"Noncausal rotary Transformer with long U-shaped skip connections predicts mel velocity conditioned on content, reference mel and speaker style. Checkpoints select an MLP or WaveNet output head.","diagram":"seed-uvit","sources":["https://github.com/Plachtaa/seed-vc"]},"flow-seed-v2-cfm":{"id":"flow-seed-v2-cfm","name":"DiT acoustic flow","role":"flow","summary":"Time-modulated rotary Transformer predicts mel velocity from ASTRAL content, target mel and CAMPPlus style. This is the CFM stage, not the upstream optional AR stage.","diagram":"seed-v2-cfm","sources":["https://github.com/Plachtaa/seed-vc"]},"encoder-coco-content-style":{"id":"encoder-coco-content-style","name":"CoCo content-style tokenizer","role":"encoder","summary":"Chromagram and Whisper features feed convolutional/ConvNeXt encoding and normalized vector quantization. This tokenizer captures content and style, not only phonetic identity.","diagram":"coco-content-style","sources":["https://huggingface.co/RMSnow/Vevo2","https://github.com/open-mmlab/Amphion/tree/main/models/svc/vevo2"]},"encoder-coco-prosody":{"id":"encoder-coco-prosody","name":"CoCo prosody tokenizer","role":"encoder","summary":"A separate chromagram-based convolutional VQ encoder represents coarse prosody. It does not use Whisper features.","diagram":"coco-prosody","sources":["https://huggingface.co/RMSnow/Vevo2"]},"head-ctc-emissions":{"id":"head-ctc-emissions","name":"CTC emission head","role":"head","summary":"Projects each acoustic frame into label log-probabilities. Forced alignment retains every frame rather than greedily collapsing labels into a transcript.","diagram":"ctc-emissions","sources":["https://huggingface.co/MahmoudAshraf/mms-300m-1130-forced-aligner"]},"encoder-wenet-conformer":{"id":"encoder-wenet-conformer","name":"WeNet Conformer","role":"encoder","summary":"Fast-U2++ convolutional subsampling and chunked relative-attention Conformer layers produce bottleneck content features, not decoded text.","diagram":"wenet-content","sources":["https://github.com/ASLP-lab/MeanVC2"]},"encoder-wavlm":{"id":"encoder-wavlm","name":"WavLM","role":"encoder","summary":"A waveform CNN and bidirectional Transformer with gated relative-position bias supply self-supervised speech features. Consumers select or mix hidden layers.","diagram":"wavlm","sources":["https://github.com/microsoft/unilm/tree/master/wavlm"]},"encoder-ecapa-features":{"id":"encoder-ecapa-features","name":"ECAPA-TDNN","role":"encoder","summary":"SE-Res2Net temporal blocks aggregate speech features and use attentive statistics pooling for a fixed speaker embedding. Here the input is WavLM features, not mel bins.","diagram":"ecapa-features","sources":["https://github.com/ASLP-lab/MeanVC2"]},"encoder-universal-timbre":{"id":"encoder-universal-timbre","name":"Universal timbre tokens","role":"encoder","summary":"Speaker-conditioned key/value tokens are queried by content bottleneck features to obtain frame-level timbre conditioning.","diagram":"meanvc-timbre","sources":["https://github.com/ASLP-lab/MeanVC2"]},"flow-meanvc-dit":{"id":"flow-meanvc-dit","name":"Mean-flow DiT","role":"flow","summary":"Time-modulated rotary Transformer blocks predict mel velocity from noise, timbre features and speaker identity. Cached context and bounded lookahead support chunked synthesis.","diagram":"meanvc-dit","sources":["https://github.com/ASLP-lab/MeanVC2"]},"frontend-linear-spectrum":{"id":"frontend-linear-spectrum","name":"Linear spectrogram","role":"dsp","summary":"Windowed STFT magnitude without mel projection.","diagram":"linear-spectrum"},"encoder-cnn-gru-speaker":{"id":"encoder-cnn-gru-speaker","name":"CNN / GRU speaker encoder","role":"encoder","summary":"Strided 2D convolutions and a GRU compress a reference spectrum into a global identity. The same encoder processes source and target recordings.","diagram":"tone-speaker","sources":["https://github.com/myshell-ai/OpenVoice/blob/main/openvoice/models.py"]},"encoder-wavenet-posterior":{"id":"encoder-wavenet-posterior","name":"WaveNet posterior encoder","role":"encoder","summary":"Gated convolutional residual/skip layers predict posterior mean and log scale from source spectra, followed by latent sampling.","diagram":"tone-posterior","sources":["https://github.com/myshell-ai/OpenVoice/blob/main/openvoice/models.py"]},"flow-tone-coupling":{"id":"flow-tone-coupling","name":"Speaker-conditioned coupling flow","role":"flow","summary":"Invertible additive coupling transforms with gated convolutional conditioners. Source-forward and target-inverse transforms use different speaker embeddings, not a diffusion timestep.","diagram":"tone-flow","sources":["https://github.com/myshell-ai/OpenVoice/blob/main/openvoice/models.py"]},"codec-hifigan-latent":{"id":"codec-hifigan-latent","name":"HiFi-GAN latent decoder","role":"codec","summary":"Continuous latent frames feed transposed-convolution upsampling and multi-receptive-field residual synthesis. Speaker conditioning depends on the owning model.","diagram":"vits-waveform","sources":["https://github.com/jaywalnut310/vits"]},"encoder-mio-content":{"id":"encoder-mio-content","name":"Local Transformer + FSQ","role":"encoder","summary":"Local rotary attention encodes WavLM content features; temporal downsampling and finite scalar quantization produce discrete content representations.","diagram":"mio-content","sources":["https://github.com/Aratako/MioCodec"]},"encoder-mio-global":{"id":"encoder-mio-global","name":"ConvNeXt speaker encoder","role":"encoder","summary":"ConvNeXt and attentive mean/std pooling turn reference WavLM features into a global voice embedding.","diagram":"mio-global","sources":["https://github.com/Aratako/MioCodec"]},"encoder-hubert-rvc":{"id":"encoder-hubert-rvc","name":"HuBERT","role":"encoder","summary":"RVC v1 selects layer 10 and a learned projection; v2 selects layer 12 without that projection. Both use waveform convolution and a bidirectional Transformer.","diagram":"hubert-rvc","sources":["https://github.com/RVC-Project/Retrieval-based-Voice-Conversion-WebUI"]},"dsp-retrieval-blend":{"id":"dsp-retrieval-blend","name":"Feature retrieval","role":"dsp","summary":"Optional nearest-neighbor training-feature lookup and blending preserve a bypass when retrieval is disabled.","diagram":"rvc-retrieval"},"encoder-rmvpe":{"id":"encoder-rmvpe","name":"RMVPE pitch encoder","role":"encoder","summary":"A residual spectrogram U-Net and bidirectional GRU estimate pitch salience, decoded to an F0 contour.","diagram":"rmvpe","sources":["https://github.com/Dream-High/RMVPE"]},"encoder-rvc-prior":{"id":"encoder-rvc-prior","name":"Relative-attention acoustic prior","role":"encoder","summary":"Content projection and optional pitch embeddings feed a relative-attention encoder that predicts latent Gaussian statistics.","diagram":"rvc-prior","sources":["https://github.com/RVC-Project/Retrieval-based-Voice-Conversion-WebUI"]},"flow-rvc-coupling":{"id":"flow-rvc-coupling","name":"Inverse coupling flow","role":"flow","summary":"The trained speaker embedding conditions a VITS-derived inverse additive-coupling stack. This is a normalizing flow, not iterative diffusion.","diagram":"vits-coupling","sources":["https://github.com/RVC-Project/Retrieval-based-Voice-Conversion-WebUI"]},"codec-rvc-nsf":{"id":"codec-rvc-nsf","name":"NSF / HiFi-GAN","role":"codec","summary":"Pitch-derived excitation is injected into residual upsampling stages alongside the converted latent and trained speaker embedding.","diagram":"rvc-nsf","sources":["https://github.com/RVC-Project/Retrieval-based-Voice-Conversion-WebUI"]},"encoder-dac-vae":{"id":"encoder-dac-vae","name":"DAC-VAE audio encoder","role":"encoder","summary":"Strided residual convolutions with Snake activations encode continuous latent means. This path does not select discrete DAC codebook tokens.","diagram":"sam-dac-encoder","sources":["https://github.com/facebookresearch/sam-audio"]},"encoder-sam-t5":{"id":"encoder-sam-t5","name":"T5 text encoder","role":"encoder","summary":"SentencePiece tokens feed a bidirectional T5 encoder with relative attention bias and ReLU feed-forward layers.","diagram":"sam-t5","sources":["https://github.com/facebookresearch/sam-audio"]},"encoder-perception-vision":{"id":"encoder-perception-vision","name":"Perception Encoder ViT","role":"encoder","summary":"Image patches pass through a vision Transformer with spatial rotary positions and learned-query attention pooling. Features condition separation, not text generation.","diagram":"sam-vision","sources":["https://github.com/facebookresearch/sam-audio"]},"encoder-temporal-anchors":{"id":"encoder-temporal-anchors","name":"Temporal anchor embeddings","role":"encoder","summary":"Positive, negative and unspecified spans become frame-aligned learned embeddings with a gated projection.","diagram":"sam-anchors","sources":["https://github.com/facebookresearch/sam-audio"]},"flow-sam-dit":{"id":"flow-sam-dit","name":"Flow-matching DiT","role":"flow","summary":"Time-modulated self-attention and text cross-attention predict a joint target/residual latent velocity field. Midpoint integration evolves noise into separated latents.","diagram":"sam-dit","sources":["https://github.com/facebookresearch/sam-audio"]},"codec-sam-dac-vae":{"id":"codec-sam-dac-vae","name":"DAC-VAE waveform decoder","role":"codec","summary":"Continuous latents feed Snake residual upsampling stages. The audio.cpp output path returns the base waveform without the retained watermark network.","diagram":"sam-dac-decoder","sources":["https://github.com/facebookresearch/sam-audio"]},"head-sano-acoustic-student":{"id":"head-sano-acoustic-student","name":"Convolutional acoustic student","role":"head","summary":"Residual convolutional duration and context networks expand phoneme features to acoustic frames. Output type depends on Nano versus PiperLite.","diagram":"sano-student","sources":["https://huggingface.co/ampixa/sanoTTS"]},"codec-sano-convnext-istft":{"id":"codec-sano-convnext-istft","name":"ConvNeXt + ISTFT decoder","role":"codec","summary":"Mel and noise projections feed ConvNeXt blocks and log-magnitude/phase prediction, followed by inverse STFT.","diagram":"sano-nano-decoder","sources":["https://huggingface.co/ampixa/sanoTTS"]},"codec-sano-residual-decoder":{"id":"codec-sano-residual-decoder","name":"Residual upsampling decoder","role":"codec","summary":"PiperLite upsamples student latents through transposed convolutions and residual banks. It does not include a VITS prior/coupling flow.","diagram":"sano-piper-decoder","sources":["https://huggingface.co/ampixa/sanoTTS"]},"frontend-supertonic-characters":{"id":"frontend-supertonic-characters","name":"Unicode character tokenizer","role":"dsp","summary":"Text normalization and language formatting followed by a Unicode character indexer, not BPE or phonemes.","diagram":"supertonic-tokenizer"},"head-supertonic-duration":{"id":"head-supertonic-duration","name":"ConvNeXt / attention duration network","role":"head","summary":"A sentence representation and duration-style tensor predict total utterance duration rather than per-phoneme alignment.","diagram":"supertonic-duration"},"encoder-supertonic-text":{"id":"encoder-supertonic-text","name":"ConvNeXt / attention text encoder","role":"encoder","summary":"Character features pass through dilated ConvNeXt and self-attention, then cross-attend to stored style tokens.","diagram":"supertonic-text","sources":["https://github.com/supertone-oss-archive/supertonic"]},"flow-supertonic-vector-field":{"id":"flow-supertonic-vector-field","name":"ConvNeXt conditional flow","role":"flow","summary":"Dilated ConvNeXt blocks alternate time conditioning and text/style attention during iterative latent generation.","diagram":"supertonic-flow","sources":["https://github.com/supertone-oss-archive/supertonic"]},"codec-supertonic-convnext":{"id":"codec-supertonic-convnext","name":"Causal ConvNeXt waveform decoder","role":"codec","summary":"Continuous latent frames are expanded and decoded to sample patches. This is not an inverse-STFT vocoder.","diagram":"supertonic-decoder","sources":["https://github.com/supertone-oss-archive/supertonic"]},"frontend-kokoro-phonemes":{"id":"frontend-kokoro-phonemes","name":"Multilingual G2P","role":"dsp","summary":"Language-specific text normalization and phonemization, then checkpoint phoneme IDs. Supplied phonemes can bypass G2P.","diagram":"phoneme-frontend","sources":["https://github.com/hexgrad/misaki","https://github.com/hexgrad/kokoro"]},"frontend-espeak-phonemes":{"id":"frontend-espeak-phonemes","name":"eSpeak phonemization","role":"dsp","summary":"Text normalization, eSpeak phonemes and checkpoint-specific symbol mapping. Token vocabularies and postprocessing are not interchangeable.","diagram":"phoneme-frontend","sources":["https://github.com/espeak-ng/espeak-ng"]},"encoder-preset-style":{"id":"encoder-preset-style","name":"Preset style lookup","role":"encoder","summary":"Select a stored voice-style vector by voice and phoneme length. No reference waveform is encoded.","diagram":"preset-style"},"encoder-styletts-text":{"id":"encoder-styletts-text","name":"CNN + BiLSTM text encoder","role":"encoder","summary":"Phoneme embeddings pass through convolutional blocks and a bidirectional LSTM, independently of PL-BERT.","diagram":"styletts-text"},"head-styletts-prosody":{"id":"head-styletts-prosody","name":"Duration / prosody predictor","role":"head","summary":"Style-conditioned recurrent layers predict durations; aligned features feed pitch/noise prediction and the acoustic decoder.","diagram":"styletts-prosody","sources":["https://github.com/hexgrad/kokoro","https://github.com/KittenML/KittenTTS"]},"codec-styletts-istft":{"id":"codec-styletts-istft","name":"Style-conditioned ISTFT decoder","role":"codec","summary":"StyleTTS2-derived residual convolutional synthesis uses harmonic/noise excitation and inverse STFT. Checkpoint sizes differ between Kokoro and Kitten.","diagram":"styletts-istft","sources":["https://github.com/hexgrad/kokoro","https://huggingface.co/KittenML/kitten-tts-mini-0.8"]},"encoder-vits-text":{"id":"encoder-vits-text","name":"VITS text encoder","role":"encoder","summary":"Phoneme embeddings, relative self-attention and convolutional feed-forward layers predict latent-prior statistics.","diagram":"vits-text","sources":["https://github.com/jaywalnut310/vits","https://github.com/rhasspy/piper"]},"head-vits-stochastic-duration":{"id":"head-vits-stochastic-duration","name":"Stochastic duration predictor","role":"head","summary":"Text-conditioned convolutional features parameterize inverse spline flows that transform duration noise into log-duration predictions.","diagram":"vits-stochastic-duration"},"head-vits-duration":{"id":"head-vits-duration","name":"Convolutional duration predictor","role":"head","summary":"Two normalized convolutional layers and a scalar projection predict log-duration without a duration sampling flow.","diagram":"vits-duration","sources":["https://huggingface.co/owensong/Inflect-Micro-v2"]},"head-vits-alignment":{"id":"head-vits-alignment","name":"Duration expansion + latent sampling","role":"head","summary":"Predicted durations expand token-level prior statistics to acoustic frames before Gaussian latent sampling.","diagram":"vits-alignment"},"flow-vits-coupling":{"id":"flow-vits-coupling","name":"VITS invertible coupling","role":"flow","summary":"Inverse residual coupling layers transform prior latents into decoder latents. This is a normalizing flow, not iterative diffusion or flow matching.","diagram":"vits-coupling","sources":["https://github.com/jaywalnut310/vits"]},"codec-vits-hifigan":{"id":"codec-vits-hifigan","name":"HiFi-GAN-style VITS decoder","role":"codec","summary":"Transposed-convolution upsampling and multi-receptive-field residual blocks decode continuous latents directly to waveform samples.","diagram":"vits-waveform","sources":["https://github.com/jaywalnut310/vits"]},"encoder-demucs-waveform":{"id":"encoder-demucs-waveform","name":"Waveform convolutional encoder","role":"encoder","summary":"HTDemucs downsamples raw audio through strided 1D convolutions and residual temporal blocks.","diagram":"demucs-encoder","sources":["https://github.com/facebookresearch/demucs/blob/main/demucs/htdemucs.py"]},"encoder-demucs-spectrum":{"id":"encoder-demucs-spectrum","name":"Spectral convolutional encoder","role":"encoder","summary":"Complex channels pass through frequency-strided convolutions, residual temporal blocks and learned frequency embeddings.","diagram":"demucs-encoder","sources":["https://github.com/facebookresearch/demucs/blob/main/demucs/htdemucs.py"]},"codec-demucs-waveform":{"id":"codec-demucs-waveform","name":"Waveform convolutional decoder","role":"codec","summary":"Add encoder skip features and upsample through transposed convolutions to predict per-stem waveform estimates.","diagram":"demucs-decoder"},"codec-demucs-spectrum":{"id":"codec-demucs-spectrum","name":"Spectral convolutional decoder","role":"codec","summary":"Restore frequency resolution and predict per-stem complex spectra using encoder skips and transposed convolutions.","diagram":"demucs-decoder"},"dsp-demucs-sum":{"id":"dsp-demucs-sum","name":"Hybrid waveform sum","role":"dsp","summary":"Undo branch normalization, reconstruct the spectral waveform and add it to the time-domain prediction for each source.","diagram":"demucs-sum"},"encoder-apollo-band-projection":{"id":"encoder-apollo-band-projection","name":"Normalized band projection","role":"encoder","summary":"Apollo concatenates normalized complex bins and a band-energy feature, then applies per-band RMSNorm and linear projection.","diagram":"apollo-bands","sources":["https://github.com/JusperLee/Apollo"]},"encoder-apollo-rope-tcn":{"id":"encoder-apollo-rope-tcn","name":"RoPE attention + TCN","role":"encoder","summary":"Alternating across-band rotary self-attention and per-band residual temporal convolutions.","diagram":"apollo-block","sources":["https://github.com/JusperLee/Apollo"]},"head-apollo-spectrum":{"id":"head-apollo-spectrum","name":"Complex spectrum prediction","role":"head","summary":"Per-band RMSNorm, linear projection and GLU directly predict real and imaginary STFT bins.","diagram":"apollo-head"},"dsp-inverse-stft":{"id":"dsp-inverse-stft","name":"Inverse STFT","role":"dsp","summary":"Inverse Fourier transform and windowed overlap-add reconstruct a waveform from its complex spectrum.","diagram":"inverse-stft"},"dsp-universr-analysis":{"id":"dsp-universr-analysis","name":"Low-band spectral analysis","role":"dsp","summary":"Resampling/bandwidth preparation and compressed complex STFT extraction retain the observed low-frequency region.","diagram":"universr-analysis"},"encoder-universr-convnext":{"id":"encoder-universr-convnext","name":"ConvNeXt conditioning encoder","role":"encoder","summary":"Frequency and sample-rate modulation plus ConvNeXt blocks encode low-band observations into temporal conditioning features.","diagram":"universr-conditioning","sources":["https://github.com/woongzip1/UniverSR"]},"flow-universr-unet":{"id":"flow-universr-unet","name":"ConvNeXt flow U-Net","role":"flow","summary":"Time-conditioned ConvNeXt U-Net estimates a complex-spectral vector field for iterative flow integration.","diagram":"universr-flow","sources":["https://github.com/woongzip1/UniverSR"]},"dsp-universr-synthesis":{"id":"dsp-universr-synthesis","name":"Band assembly + ISTFT","role":"dsp","summary":"Retain observed low bins, insert generated high bins, undo spectral compression and reconstruct the waveform.","diagram":"universr-synthesis"},"encoder-roformer-band-split":{"id":"encoder-roformer-band-split","name":"Band-wise projection","role":"encoder","summary":"Normalize and linearly project real/imaginary bins within each band. Band overlap differs between BS and Mel-Band RoFormer.","diagram":"roformer-bands","sources":["https://github.com/lucidrains/BS-RoFormer"]},"head-roformer-mask":{"id":"head-roformer-mask","name":"Band-wise mask MLP","role":"head","summary":"Band-specific MLPs and GLU output complex masks; overlapping bins are averaged for Mel-Band RoFormer.","diagram":"roformer-mask"},"dsp-rnnoise-analysis":{"id":"dsp-rnnoise-analysis","name":"Band / pitch analysis","role":"dsp","summary":"Windowed Fourier analysis, band cepstra and pitch-correlation features for RNNoise.","diagram":"rnnoise-analysis","sources":["https://github.com/xiph/rnnoise"]},"encoder-rnnoise-conv-gru":{"id":"encoder-rnnoise-conv-gru","name":"Causal CNN + GRU","role":"encoder","summary":"Two causal convolutions and three GRUs. Concatenated intermediate states feed sigmoid gain and VAD heads.","diagram":"rnnoise-network","sources":["https://github.com/xiph/rnnoise"]},"dsp-rnnoise-synthesis":{"id":"dsp-rnnoise-synthesis","name":"Pitch filter + band gains","role":"dsp","summary":"Pitch filtering, smoothed band gains and overlap-add synthesis retain the signal-processing character of RNNoise.","diagram":"rnnoise-synthesis"},"dsp-deepfilter-analysis":{"id":"dsp-deepfilter-analysis","name":"ERB + complex features","role":"dsp","summary":"Normalized ERB-band energies and normalized low-frequency complex bins form two feature branches.","diagram":"deepfilter-analysis","sources":["https://github.com/Rikorose/DeepFilterNet"]},"encoder-deepfilter-conv-gru":{"id":"encoder-deepfilter-conv-gru","name":"CNN + grouped GRU","role":"encoder","summary":"DeepFilterNet2 combines convolutional feature branches with grouped recurrent layers and separate gain/filter decoders.","diagram":"deepfilter-network","sources":["https://github.com/Rikorose/DeepFilterNet/blob/main/DeepFilterNet/df/deepfilternet2.py"]},"head-deepfilter-synthesis":{"id":"head-deepfilter-synthesis","name":"Complex deep filtering","role":"head","summary":"ERB gains attenuate the spectrum; predicted complex coefficients filter low-frequency bins across frames. Inverse STFT reconstructs speech.","diagram":"deepfilter-synthesis"},"dsp-complex-stft":{"id":"dsp-complex-stft","name":"Complex STFT","role":"dsp","summary":"Windowing and Fourier transform retain both real and imaginary spectral channels.","diagram":"complex-stft"},"head-complex-mask-synthesis":{"id":"head-complex-mask-synthesis","name":"Complex mask + ISTFT","role":"head","summary":"Complex multiplication of the predicted mask and input spectrum, followed by inverse STFT.","diagram":"complex-mask-synthesis"},"dsp-zipenhancer-analysis":{"id":"dsp-zipenhancer-analysis","name":"Compressed spectral features","role":"dsp","summary":"Energy normalization and STFT followed by magnitude compression and complex-feature reconstruction.","diagram":"zipenhancer-analysis"},"head-magnitude-phase-synthesis":{"id":"head-magnitude-phase-synthesis","name":"Magnitude / phase + ISTFT","role":"head","summary":"Independent magnitude and phase branches reconstruct a complex spectrum for inverse STFT.","diagram":"magnitude-phase-synthesis"},"encoder-flashsr-residual-cnn":{"id":"encoder-flashsr-residual-cnn","name":"Residual upsampling CNN","role":"encoder","summary":"Linear 3x upsampling and parallel dilated residual branches with alias-filtered periodic activations.","diagram":"flashsr","sources":["https://github.com/ysharma3501/FlashSR"]},"head-nemotron-speaker-upsampling":{"id":"head-nemotron-speaker-upsampling","name":"Upsampled speaker head","role":"head","summary":"Temporal convolution and subpixel upsampling restore frame resolution. Independent sigmoid outputs permit overlapping speakers.","diagram":"nemotron-diar-head"},"encoder-sortformer-transformer":{"id":"encoder-sortformer-transformer","name":"Post-norm Transformer","role":"encoder","summary":"Sortformer's second encoder stack uses post-residual LayerNorm and ReLU feed-forward layers, distinct from the preceding Conformer.","diagram":"sortformer-transformer"},"encoder-sortformer-streaming-conformer":{"id":"encoder-sortformer-streaming-conformer","name":"FastConformer + speaker memory","role":"encoder","summary":"Subsampled chunk embeddings are combined with arrival-order speaker cache and FIFO context before Conformer processing.","diagram":"sortformer-streaming","sources":["https://huggingface.co/nvidia/diar_streaming_sortformer_4spk-v2.1"]},"head-sortformer":{"id":"head-sortformer","name":"Speaker activity head","role":"head","summary":"ReLU projections and independent sigmoid outputs predict arrival-ordered speaker activity, including overlap.","diagram":"sortformer-head"},"dsp-speaker-segmentation":{"id":"dsp-speaker-segmentation","name":"Speaker segmentation","role":"dsp","summary":"Converts speaker activity probabilities into timed turns using model-specific thresholds and interval processing. It does not recognize words.","diagram":"speaker-segmentation"},"encoder-qwen3-align-backbone":{"id":"encoder-qwen3-align-backbone","name":"Qwen3 causal backbone","role":"encoder","summary":"Qwen3 decoder blocks evaluated over the supplied audio/transcript prompt for classification. This route does not run an autoregressive token-generation loop.","diagram":"qwen3-layer","sources":["https://huggingface.co/Qwen/Qwen3-ForcedAligner-0.6B"]},"frontend-qwen-align-prompt":{"id":"frontend-qwen-align-prompt","name":"Word / timestamp prompt","role":"dsp","summary":"Transcript normalization and tokenization insert two timestamp placeholders per word alongside audio slots.","diagram":"qwen-align-prompt"},"frontend-mms-transcript":{"id":"frontend-mms-transcript","name":"Transcript normalization","role":"dsp","summary":"Normalizes and maps the supplied transcript to the aligner's character vocabulary.","diagram":"mms-text"},"dsp-stft-magnitude":{"id":"dsp-stft-magnitude","name":"Magnitude spectrum","role":"dsp","summary":"Windowed Fourier analysis produces magnitude features, not mel-filterbank features.","diagram":"stft-magnitude"},"head-silero-probability":{"id":"head-silero-probability","name":"Sigmoid speech head","role":"head","summary":"ReLU and a one-channel projection turn the recurrent hidden state into speech probability.","diagram":"silero-head"},"head-binary-speech-classifier":{"id":"head-binary-speech-classifier","name":"Speech / non-speech classifier","role":"head","summary":"A learned projection followed by probability normalization distinguishes speech from non-speech.","diagram":"binary-speech-head"},"dsp-vad-segmentation":{"id":"dsp-vad-segmentation","name":"Speech interval extraction","role":"dsp","summary":"Thresholding, duration constraints and optional padding convert probabilities into intervals. Policies are model-specific.","diagram":"vad-segmentation"},"frontend-moonshine-time-domain":{"id":"frontend-moonshine-time-domain","name":"Time-domain frontend","role":"encoder","summary":"Normalized waveform frames with asinh compression, linear projection and causal strided convolutions. No spectrogram is computed.","diagram":"moonshine-frontend","sources":["https://huggingface.co/moonshine-ai/moonshine-streaming-tiny"]},"encoder-moonshine-windowed":{"id":"encoder-moonshine-windowed","name":"Sliding-window Transformer","role":"encoder","summary":"Moonshine Streaming's local-attention encoder does not use positional embeddings inside the attention stack.","diagram":"moonshine-encoder"},"encoder-moonshine-position-adapter":{"id":"encoder-moonshine-position-adapter","name":"Positional adapter","role":"encoder","summary":"Adds learned absolute positions after acoustic encoding and aligns the decoder memory width.","diagram":"moonshine-adapter"},"ar-moonshine-cross-attention":{"id":"ar-moonshine-cross-attention","name":"AR cross-attention Transformer","role":"ar","summary":"Moonshine uses RoPE causal self-attention, audio cross-attention and a gated SiLU feed-forward sublayer, with LayerNorm.","diagram":"moonshine-decoder"},"encoder-hviske-conformer":{"id":"encoder-hviske-conformer","name":"Conformer","role":"encoder","summary":"Depthwise Conv2D subsampling, relative-position Conformer blocks and a projection to the text decoder width.","diagram":"fastconformer","sources":["https://huggingface.co/syvai/hviske-v5.3"]},"encoder-audio8-mlp-adapter":{"id":"encoder-audio8-mlp-adapter","name":"Residual MLP + pooling adapter","role":"encoder","summary":"Audio8 stacks normalized residual MLPs, pools over time and projects acoustic features into Qwen2's embedding width.","diagram":"audio8-adapter","sources":["https://huggingface.co/Edge0/Audio8-ASR-0.1B"]},"frontend-kaldi-fbank":{"id":"frontend-kaldi-fbank","name":"Kaldi-style filterbank","role":"dsp","summary":"Windowed, pre-emphasized audio is transformed into log-mel filterbank energies. Framing and normalization follow the checkpoint.","diagram":"kaldi-fbank"},"head-stateless-transducer":{"id":"head-stateless-transducer","name":"Stateless transducer","role":"head","summary":"Kroko's two-token grouped-convolution predictor meets acoustic features in a tanh joiner. Unlike an LSTM predictor, it retains token context rather than recurrent hidden state.","diagram":"stateless-transducer"},"encoder-granite-conformer":{"id":"encoder-granite-conformer","name":"Self-conditioned Conformer","role":"encoder","summary":"Granite TurboCTC's block-attention Conformer feeds an intermediate CTC distribution back into its hidden representation.","diagram":"granite-encoder","sources":["https://huggingface.co/ibm-granite/granite-speech-5.0-470m-turboctc"]},"encoder-higgs-stt-projector":{"id":"encoder-higgs-stt-projector","name":"Temporal convolution + MLP adapter","role":"encoder","summary":"Higgs STT reduces audio time resolution and maps encoder features into Qwen3's embedding width.","diagram":"higgs-stt-projector","sources":["https://huggingface.co/bosonai/higgs-audio-v3-stt"]},"encoder-moss-stt-projector":{"id":"encoder-moss-stt-projector","name":"Frame stacking + MLP adapter","role":"encoder","summary":"MOSS groups Whisper frames before SiLU MLP projection and LayerNorm. Timestamp markers are inserted into the text prompt separately.","diagram":"moss-stt-projector","sources":["https://huggingface.co/OpenMOSS-Team/MOSS-Transcribe-Diarize"]},"encoder-fun-transformer-adapter":{"id":"encoder-fun-transformer-adapter","name":"Transformer adapter","role":"encoder","summary":"Fun-ASR-Nano projects acoustic features and applies two self-attention adapter layers before its Qwen3 decoder.","diagram":"fun-adapter","sources":["https://huggingface.co/FunAudioLLM/Fun-ASR-Nano-2512/blob/main/config.yaml"]},"frontend-fbank-lfr":{"id":"frontend-fbank-lfr","name":"Filterbank + LFR / CMVN","role":"dsp","summary":"Filterbank extraction, low-frame-rate stacking and checkpoint normalization for SenseVoice-family acoustic features.","diagram":"fbank-lfr","sources":["https://github.com/FunAudioLLM/SenseVoice"]},"encoder-language-fusion":{"id":"encoder-language-fusion","name":"Language conditioning projection","role":"encoder","summary":"Nemotron 3.5 broadcasts a language-ID vector over time, concatenates it with acoustic features, and projects the result for the transducer.","diagram":"language-fusion","sources":["https://huggingface.co/nvidia/nemotron-3.5-asr-streaming-0.6b"]},"frontend-logmel":{"id":"frontend-logmel","name":"Log-mel frontend","role":"dsp","summary":"Windowed spectral features, mel filtering and checkpoint-specific normalization.","diagram":"mel-features","sources":["https://github.com/NVIDIA/NeMo/blob/main/nemo/collections/asr/parts/preprocessing/features.py"]},"encoder-gigaam-conformer":{"id":"encoder-gigaam-conformer","name":"Rotary Conformer","role":"encoder","summary":"GigaAM's Conv1D subsampling and rotary Conformer. It uses an unusual pre-projection rotary transform for Q/K.","diagram":"gigaam-conformer","sources":["https://github.com/salute-developers/GigaAM"]},"encoder-cohere-conformer":{"id":"encoder-cohere-conformer","name":"Conformer","role":"encoder","summary":"Cohere's subsampled Conformer audio tower with relative-position attention.","diagram":"fastconformer","sources":["https://huggingface.co/CohereLabs/cohere-transcribe-03-2026"]},"ar-nemo-cross-attention":{"id":"ar-nemo-cross-attention","name":"AR cross-attention Transformer","role":"ar","summary":"Causal text decoding with cross-attention to encoded audio. Canary and Cohere have separate vocabularies, widths and weights.","diagram":"nemo-decoder","sources":["https://huggingface.co/nvidia/canary-180m-flash","https://huggingface.co/CohereLabs/cohere-transcribe-03-2026"]},"ar-qwen3-code-predictor":{"id":"ar-qwen3-code-predictor","name":"Qwen3 depth AR","role":"ar","summary":"A smaller Qwen3 decoder autoregressively fills remaining codebooks within each audio frame, not successive time frames.","diagram":"qwen3-depth","sources":["https://github.com/QwenLM/Qwen3-TTS"]},"encoder-qwen-ecapa":{"id":"encoder-qwen-ecapa","name":"ECAPA-TDNN","role":"encoder","summary":"Mel features, SE-Res2Net blocks and attentive statistics pooling produce the Base model's speaker embedding.","diagram":"qwen3-speaker","sources":["https://github.com/QwenLM/Qwen3-TTS"]},"encoder-qwen-speech-codes":{"id":"encoder-qwen-speech-codes","name":"Mimi-derived speech encoder","role":"encoder","summary":"Reference audio codes are used for in-context cloning. The speaker-embedding-only path omits this branch.","diagram":"breeze-reference","sources":["https://github.com/QwenLM/Qwen3-TTS"]},"frontend-bpe":{"id":"frontend-bpe","name":"BPE tokenizer","role":"dsp","summary":"Checkpoint-specific vocabulary, normalization and special tokens. This label does not mean interchangeable vocabularies.","diagram":"bpe","sources":["https://huggingface.co/docs/tokenizers/api/models#tokenizers.models.BPE"]},"frontend-soprano":{"id":"frontend-soprano","name":"Normalization + BPE","role":"dsp","summary":"Soprano text normalization followed by prompt tokenization.","diagram":"soprano-text","sources":["https://github.com/ekwek1/soprano"]},"frontend-qwen-mel":{"id":"frontend-qwen-mel","name":"Log-mel frontend","role":"dsp","summary":"16 kHz audio and 128 mel bins. Whisper-style features, not Whisper encoder weights.","diagram":"logmel","sources":["https://github.com/QwenLM/Qwen3-ASR"]},"encoder-higgs-reference":{"id":"encoder-higgs-reference","name":"HuBERT + acoustic encoder","role":"encoder","summary":"Higgs combines semantic and acoustic reference features and quantizes them into prompt codebooks.","diagram":"higgs-reference","sources":["https://huggingface.co/bosonai/higgs-tts-3-4b","https://github.com/0xShug0/audio.cpp/blob/main/src/models/higgs_audio_tts/codec.cpp"]},"encoder-mio-reference":{"id":"encoder-mio-reference","name":"WavLM + global encoder","role":"encoder","summary":"Reference voice features condition MioCodec waveform synthesis, not the MioTTS language model.","diagram":"mio-reference","sources":["https://github.com/Aratako/MioCodec","https://github.com/0xShug0/audio.cpp/blob/main/src/models/miotts/session.cpp"]},"encoder-vibe-reference":{"id":"encoder-vibe-reference","name":"Acoustic reference tokenizer","role":"encoder","summary":"Continuous reference latents are projected into the Qwen2 input space.","diagram":"vibe-reference","sources":["https://github.com/microsoft/VibeVoice","https://huggingface.co/microsoft/VibeVoice-1.5B/blob/main/config.json"]}},"models":[{"id":"higgs_audio_tts","family":"higgs_audio_tts","name":"Higgs Audio v3 TTS","task":"speech-synthesis","tasks":["tts","clone"],"summary":"Qwen3-based multi-codebook speech generation. Reference audio becomes an optional AR prompt; the optional transcript supplies textual context.","routes":[{"name":"TTS / voice cloning","nodes":[{"id":"text","kind":"input","label":"Text + optional transcript"},{"id":"ref","kind":"input","label":"Reference audio","optional":true},{"id":"tok","component":"frontend-bpe"},{"id":"enc","component":"encoder-higgs-reference"},{"id":"ar","component":"ar-qwen3","detail":"4B-class, 36 layers. Interleaved text and 8-codebook audio prompt."},{"id":"codec","component":"codec-higgs-codec"},{"id":"out","kind":"output","label":"24 kHz speech"}],"edges":[{"from":"text","to":"tok"},{"from":"ref","to":"enc","optional":true},{"from":"tok","to":"ar","label":"Text IDs"},{"from":"enc","to":"ar","label":"Reference codes","optional":true},{"from":"ar","to":"codec","label":"Delayed codebooks"},{"from":"codec","to":"out"}]}],"sources":["https://huggingface.co/bosonai/higgs-tts-3-4b"],"packages":[{"id":"higgs_audio_tts_4b_q8_0","display_name":"Higgs Audio v3 TTS 4B Q8_0 GGUF","precision":"q8_0"},{"id":"higgs_audio_tts_4b_bf16","display_name":"Higgs Audio v3 TTS 4B BF16 GGUF","precision":"bf16"}],"docs":["docs/models/higgs_audio_tts.md","docs/tts.md","docs/gguf.md"],"variants":["v3 TTS 4B"],"inputs":[{"type":"text","label":"Text","required":true},{"type":"audio","label":"Reference voice","required":false},{"type":"text","label":"Reference transcript","required":false}],"detailStatus":"Reference branches audited.","usageDoc":"docs/models/higgs_audio_tts.md"},{"id":"soprano_tts","family":"soprano_tts","name":"Soprano","task":"speech-synthesis","tasks":["tts"],"summary":"Compact Qwen3 speech generator with Vocos synthesis. Generated hidden features drive the vocoder; this packaged path does not take a reference voice.","routes":[{"name":"Text to speech","nodes":[{"id":"text","kind":"input","label":"Text"},{"id":"tok","component":"frontend-soprano"},{"id":"ar","component":"ar-qwen3","detail":"80M model; token sampling and hidden-feature generation."},{"id":"codec","component":"codec-vocos","detail":"Consumes hidden frames, not codec codebook IDs."},{"id":"out","kind":"output","label":"32 kHz speech"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"ar"},{"from":"ar","to":"codec","label":"Hidden features"},{"from":"codec","to":"out"}]}],"sources":["https://huggingface.co/ekwek/Soprano-1.1-80M"],"packages":[{"id":"soprano_1_1_80m_q8_0","display_name":"Soprano-1.1-80M Q8_0 GGUF","precision":"q8_0"},{"id":"soprano_1_1_80m_bf16","display_name":"Soprano-1.1-80M BF16 GGUF","precision":"bf16"}],"docs":["docs/community_models/soprano_tts.md"],"variants":["1.1 80M"],"inputs":[{"type":"text","label":"Text","required":true}],"detailStatus":"Hidden-feature synthesis audited.","usageDoc":"docs/community_models/soprano_tts.md"},{"id":"miotts","family":"miotts","name":"MioTTS","task":"speech-synthesis","tasks":["tts","clone"],"summary":"Qwen3 generates content tokens from text. WavLM and the global reference encoder supply speaker conditioning to MioCodec waveform reconstruction. Other upstream MioTTS backbones are not implied by this package.","routes":[{"name":"Zero-shot voice cloning","nodes":[{"id":"text","kind":"input","label":"Text"},{"id":"ref","kind":"input","label":"Reference audio"},{"id":"tok","component":"frontend-bpe"},{"id":"enc","component":"encoder-mio-reference"},{"id":"ar","component":"ar-qwen3","detail":"Qwen3-1.7B-Base initialization; content-token generation."},{"id":"codec","component":"codec-miocodec"},{"id":"out","kind":"output","label":"Speech waveform"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"ar"},{"from":"ref","to":"enc"},{"from":"enc","to":"codec","label":"Global voice embedding"},{"from":"ar","to":"codec","label":"Content tokens"},{"from":"codec","to":"out"}]}],"sources":["https://huggingface.co/Aratako/MioTTS-1.7B"],"packages":[{"id":"miotts_1_7b_q8_0","display_name":"MioTTS 1.7B Q8_0 GGUF","precision":"q8_0"},{"id":"miotts_1_7b_bf16","display_name":"MioTTS 1.7B BF16 GGUF","precision":"bf16"},{"id":"miotts_1_7b_orig","display_name":"MioTTS 1.7B Original-Dtype GGUF","precision":"orig"}],"docs":["docs/models/miotts.md","docs/tts.md","docs/gguf.md"],"variants":["1.7B"],"inputs":[{"type":"text","label":"Text","required":true},{"type":"audio","label":"Reference voice","required":true}],"detailStatus":"Reference conditioning enters the codec, not the LM.","usageDoc":"docs/models/miotts.md"},{"id":"neutts","family":"neutts","name":"NeuTTS","task":"speech-synthesis","tasks":["tts"],"summary":"NeuTTS-2E uses a compact Qwen3 AR backbone to continue a stored speaker's text/audio-code prompt. Emotion is encoded in the prompt. NeuCodec converts the generated finite-scalar-quantized tokens through a Transformer/convolutional spectral decoder to speech.","routes":[{"name":"Preset speaker / emotional TTS","nodes":[{"id":"text","kind":"input","label":"Text + emotion"},{"id":"voice","kind":"input","label":"Speaker ID"},{"id":"preset","component":"encoder-neutts-preset"},{"id":"tokenizer","component":"frontend-bpe"},{"id":"ar","component":"ar-qwen3"},{"id":"codec","component":"codec-neucodec"},{"id":"out","kind":"output","label":"24 kHz speech"}],"edges":[{"from":"voice","to":"preset"},{"from":"text","to":"tokenizer"},{"from":"preset","to":"tokenizer","label":"Reference transcript"},{"from":"preset","to":"ar","label":"Stored speech codes"},{"from":"tokenizer","to":"ar"},{"from":"ar","to":"codec"},{"from":"codec","to":"out"}]}],"sources":["https://huggingface.co/neuphonic/neutts-2e"],"packages":[{"id":"neutts_2e_orig","display_name":"NeuTTS 2E Original-Precision GGUF","precision":"orig"}],"docs":["docs/tts.md","docs/models/neutts.md","docs/gguf.md"],"variants":["NeuTTS 2E"],"inputs":[{"type":"text","label":"Target text","required":true},{"type":"id","label":"Packaged speaker","required":false},{"type":"control","label":"Emotion","required":false}],"detailStatus":"Packaged speaker conditioning and codec blocks audited; no user reference-waveform branch in this route.","usageDoc":"docs/models/neutts.md"},{"id":"fireredtts3","family":"fireredtts3","name":"FireRedTTS3","task":"speech-synthesis","tasks":["tts","clone","design"],"summary":"Qwen3 predicts continuous speech patches through a small conditional flow DiT. RedAE encodes references and reconstructs audio. Base adds CAM++ speaker identity; Instruct instead embeds reference/source patches in its multimodal prompt and can generate text before audio.","routes":[{"name":"Base / voice cloning","inputs":[{"type":"text","label":"Speech text + reference transcript + language","required":true},{"type":"audio","label":"Reference voice","required":true}],"nodes":[{"id":"text","kind":"input","label":"Text + reference transcript + language"},{"id":"ref","kind":"input","label":"Reference voice"},{"id":"tok","component":"frontend-bpe","detail":"Local text normalization and language-tagged prompt."},{"id":"speaker","component":"encoder-campplus"},{"id":"enc","component":"encoder-redae"},{"id":"patch","component":"encoder-firered-patch"},{"id":"ar","component":"ar-qwen3"},{"id":"flow","component":"flow-dit-flow","detail":"Base concatenates projected CAM++ speaker conditioning."},{"id":"codec","component":"codec-redae"},{"id":"out","kind":"output","label":"24 kHz speech"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"ar"},{"from":"ref","to":"speaker"},{"from":"ref","to":"enc"},{"from":"enc","to":"patch"},{"from":"patch","to":"ar"},{"from":"speaker","to":"ar","label":"Projected identity"},{"from":"speaker","to":"flow","label":"Projected identity"},{"from":"ar","to":"flow","label":"Hidden-state conditioning"},{"from":"flow","to":"codec","label":"Continuous latent patches"},{"from":"codec","to":"out"}]},{"name":"Instruct / reference cloning","inputs":[{"type":"text","label":"Speech text / instruction","required":true},{"type":"audio","label":"Reference voice","required":true}],"nodes":[{"id":"text","kind":"input","label":"Speech text / instruction"},{"id":"ref","kind":"input","label":"Reference voice"},{"id":"tok","component":"frontend-bpe"},{"id":"enc","component":"encoder-redae"},{"id":"patch","component":"encoder-firered-patch"},{"id":"ar","component":"ar-qwen3"},{"id":"flow","component":"flow-dit-flow","detail":"Instruct has no separate CAM++ speaker input."},{"id":"codec","component":"codec-redae"},{"id":"out","kind":"output","label":"24 kHz speech"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"ar"},{"from":"ref","to":"enc"},{"from":"enc","to":"patch"},{"from":"patch","to":"ar"},{"from":"ar","to":"flow"},{"from":"flow","to":"codec"},{"from":"codec","to":"out"}]},{"name":"Instruct / voice design","inputs":[{"type":"text","label":"Speech text + voice description","required":true}],"nodes":[{"id":"text","kind":"input","label":"Speech text + voice description"},{"id":"tok","component":"frontend-bpe"},{"id":"ar","component":"ar-qwen3","detail":"Text planning precedes acoustic patch generation."},{"id":"plan","kind":"output","label":"Voice plan text"},{"id":"flow","component":"flow-dit-flow"},{"id":"codec","component":"codec-redae"},{"id":"out","kind":"output","label":"24 kHz speech"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"ar"},{"from":"ar","to":"plan"},{"from":"ar","to":"flow"},{"from":"flow","to":"codec"},{"from":"codec","to":"out"}]},{"name":"Instruct / semantic or acoustic editing","inputs":[{"type":"text","label":"Edit instruction","required":true},{"type":"audio","label":"Source recording","required":true}],"nodes":[{"id":"text","kind":"input","label":"Edit instruction"},{"id":"ref","kind":"input","label":"Source recording"},{"id":"tok","component":"frontend-bpe"},{"id":"enc","component":"encoder-redae"},{"id":"patch","component":"encoder-firered-patch"},{"id":"ar","component":"ar-qwen3","detail":"Semantic editing generates revised text before acoustic generation; acoustic editing can skip text generation."},{"id":"flow","component":"flow-dit-flow"},{"id":"codec","component":"codec-redae"},{"id":"out","kind":"output","label":"Edited speech"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"ar"},{"from":"ref","to":"enc"},{"from":"enc","to":"patch"},{"from":"patch","to":"ar"},{"from":"ar","to":"flow"},{"from":"flow","to":"codec"},{"from":"codec","to":"out"}]}],"sources":["https://huggingface.co/FireRedTeam/FireRedTTS3"],"packages":[{"id":"fireredtts3_instruct_orig","display_name":"FireRedTTS3 Instruct Orig GGUF","precision":"orig"},{"id":"fireredtts3_instruct_q8_0","display_name":"FireRedTTS3 Instruct Q8_0 GGUF","precision":"q8_0"},{"id":"fireredtts3_base_orig","display_name":"FireRedTTS3 Base Orig GGUF","precision":"orig"},{"id":"fireredtts3_base_q8_0","display_name":"FireRedTTS3 Base Q8_0 GGUF","precision":"q8_0"}],"docs":["docs/models/fireredtts3.md","docs/gguf.md"],"variants":["Base","Instruct / VoiceDesign"],"inputs":[{"type":"text","label":"Speech text / instruction","required":true},{"type":"audio","label":"Reference or editing source, depending on route","required":false}],"detailStatus":"Base, Instruct cloning, design and editing branches audited against upstream and the current session.","usageDoc":"docs/models/fireredtts3.md"},{"id":"qwen3_tts","family":"qwen3_tts","name":"Qwen3-TTS","task":"speech-synthesis","tasks":["tts","clone","design"],"summary":"A temporal Qwen3 talker generates the first codebook; a smaller Qwen3 code predictor fills the remaining codebooks for each frame. Base uses reference voice conditioning, VoiceDesign uses a written description, and CustomVoice uses packaged speaker embeddings.","routes":[{"name":"Base / voice cloning","inputs":[{"type":"text","label":"Text","required":true},{"type":"audio","label":"Reference voice","required":true},{"type":"text","label":"Reference transcript for ICL","required":false}],"nodes":[{"id":"text","kind":"input","label":"Text + optional transcript"},{"id":"ref","kind":"input","label":"Reference audio"},{"id":"tok","component":"frontend-bpe"},{"id":"speaker","component":"encoder-qwen-ecapa"},{"id":"refcodes","component":"encoder-qwen-speech-codes"},{"id":"ar","component":"ar-qwen3-talker"},{"id":"depth","component":"ar-qwen3-code-predictor"},{"id":"codec","component":"codec-qwen3-speech-tokenizer"},{"id":"out","kind":"output","label":"Speech waveform"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"ar"},{"from":"ref","to":"speaker"},{"from":"speaker","to":"ar","label":"Speaker embedding"},{"from":"ref","to":"refcodes","optional":true},{"from":"refcodes","to":"ar","label":"ICL audio prompt","optional":true},{"from":"ar","to":"depth","label":"Hidden state + first code"},{"from":"depth","to":"codec","label":"Complete codebooks"},{"from":"codec","to":"out"}]},{"name":"VoiceDesign","inputs":[{"type":"text","label":"Text","required":true},{"type":"text","label":"Voice description","required":true}],"nodes":[{"id":"text","kind":"input","label":"Text + voice description"},{"id":"tok","component":"frontend-bpe"},{"id":"ar","component":"ar-qwen3-talker"},{"id":"depth","component":"ar-qwen3-code-predictor"},{"id":"codec","component":"codec-qwen3-speech-tokenizer"},{"id":"out","kind":"output","label":"Speech waveform"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"ar"},{"from":"ar","to":"depth","label":"Hidden state + first code"},{"from":"depth","to":"codec"},{"from":"codec","to":"out"}]},{"name":"CustomVoice","inputs":[{"type":"text","label":"Text","required":true},{"type":"id","label":"Packaged speaker","required":true},{"type":"text","label":"Style instruction","required":false}],"nodes":[{"id":"text","kind":"input","label":"Text + optional style"},{"id":"voice","kind":"input","label":"Packaged speaker ID"},{"id":"tok","component":"frontend-bpe"},{"id":"ar","component":"ar-qwen3-talker"},{"id":"depth","component":"ar-qwen3-code-predictor"},{"id":"codec","component":"codec-qwen3-speech-tokenizer"},{"id":"out","kind":"output","label":"Speech waveform"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"ar"},{"from":"voice","to":"ar","label":"Speaker embedding"},{"from":"ar","to":"depth","label":"Hidden state + first code"},{"from":"depth","to":"codec"},{"from":"codec","to":"out"}]}],"sources":["https://huggingface.co/Qwen/Qwen3-TTS-12Hz-1.7B-Base"],"packages":[{"id":"qwen3_tts_1_7b_base_q8_0","display_name":"Qwen3 TTS 12Hz 1.7B Base Q8_0 GGUF","precision":"q8_0"},{"id":"qwen3_tts_1_7b_base_bf16","display_name":"Qwen3 TTS 12Hz 1.7B Base BF16 GGUF","precision":"bf16"},{"id":"qwen3_tts_1_7b_base_orig","display_name":"Qwen3 TTS 12Hz 1.7B Base Original-Dtype GGUF","precision":"orig"},{"id":"qwen3_tts_1_7b_customvoice_q8_0","display_name":"Qwen3 TTS 12Hz 1.7B CustomVoice Q8_0 GGUF","precision":"q8_0"},{"id":"qwen3_tts_1_7b_customvoice_bf16","display_name":"Qwen3 TTS 12Hz 1.7B CustomVoice BF16 GGUF","precision":"bf16"},{"id":"qwen3_tts_1_7b_voicedesign_q8_0","display_name":"Qwen3 TTS 12Hz 1.7B VoiceDesign Q8_0 GGUF","precision":"q8_0"},{"id":"qwen3_tts_1_7b_voicedesign_bf16","display_name":"Qwen3 TTS 12Hz 1.7B VoiceDesign BF16 GGUF","precision":"bf16"},{"id":"qwen3_tts_0_6b_base_q8_0","display_name":"Qwen3 TTS 12Hz 0.6B Base Q8_0 GGUF","precision":"q8_0"},{"id":"qwen3_tts_0_6b_base_bf16","display_name":"Qwen3 TTS 12Hz 0.6B Base BF16 GGUF","precision":"bf16"},{"id":"qwen3_tts_1_7b_base_safetensors","display_name":"Qwen3 TTS 12Hz 1.7B Base Safetensors","precision":"native"}],"docs":["docs/models/qwen3.md","docs/tts.md","docs/gguf.md"],"variants":["12Hz 1.7B Base","12Hz 0.6B Base","12Hz 1.7B VoiceDesign","12Hz 1.7B CustomVoice"],"inputs":[{"type":"text","label":"Text","required":true}],"detailStatus":"Variant-specific conditioning, speaker encoder, depth AR and speech codec audited.","usageDoc":"docs/models/qwen3.md#qwen3-tts-base"},{"id":"vieneu_v3_turbo","family":"vieneu_v3_turbo","name":"VieNeu-TTS v3 Turbo","task":"speech-synthesis","tasks":["tts","clone"],"summary":"VieNeu uses a trained-from-scratch rotary temporal AR talker and a separate within-frame acoustic decoder. MOSS Nano encodes references and reconstructs 48 kHz audio. Speaker identity can come from a preset or supplied embedding; the upstream CAM++ encoder is not part of the packaged AudioCPP path.","routes":[{"name":"v3 Turbo / preset or reference conditioning","nodes":[{"id":"text","kind":"input","label":"Text or prepared phonemes"},{"id":"ref","kind":"input","label":"Reference recording","optional":true},{"id":"prepared","kind":"input","label":"Prepared reference codes","optional":true},{"id":"identity","kind":"input","label":"Preset / speaker embedding","optional":true},{"id":"front","component":"frontend-vieneu-phones"},{"id":"enc","component":"encoder-moss-reference","detail":"MOSS Nano; mono reference duplicated into codec channels."},{"id":"speaker","component":"encoder-vieneu-anchor"},{"id":"ar","component":"ar-vieneu-talker"},{"id":"depth","component":"ar-code-predictor"},{"id":"codec","component":"codec-moss-audio-tokenizer","detail":"MOSS Nano 48 kHz reconstruction, mixed to mono for speech output."},{"id":"out","kind":"output","label":"48 kHz speech"}],"edges":[{"from":"text","to":"front"},{"from":"front","to":"ar"},{"from":"ref","to":"enc","optional":true},{"from":"enc","to":"ar","optional":true,"label":"Prompt codebooks"},{"from":"prepared","to":"ar","optional":true},{"from":"identity","to":"speaker","optional":true},{"from":"speaker","to":"ar","optional":true,"label":"Additive speaker anchor"},{"from":"ar","to":"depth","label":"Temporal state"},{"from":"depth","to":"codec","label":"16 codebooks per frame"},{"from":"codec","to":"out"}]}],"sources":["https://huggingface.co/pnnbao-ump/VieNeu-TTS-v3-Turbo"],"packages":[{"id":"vieneu_v3_turbo_q8_0","display_name":"VieNeu-TTS v3 Turbo Q8_0 GGUF","precision":"q8_0"},{"id":"vieneu_v3_turbo_bf16","display_name":"VieNeu-TTS v3 Turbo BF16 GGUF","precision":"bf16"},{"id":"vietneu_tts_v3_turbo_q8_0","display_name":"VieNeu-TTS v3 Turbo Q8_0 GGUF (first community port, July 2026 weights)","precision":"q8_0"}],"docs":["docs/community_models/vieneu_v3_turbo.md","docs/tts.md","docs/gguf.md"],"variants":["v3 Turbo"],"inputs":[{"type":"text","label":"Phonemes, or text with SEA-G2P configured","required":true},{"type":"audio","label":"Reference recording or prepared reference codes","required":false},{"type":"embedding","label":"Preset or supplied speaker embedding","required":false}],"detailStatus":"Current session and checkpoint architecture audited; obsolete Qwen speech-codec route removed.","usageDoc":"docs/community_models/vieneu_v3_turbo.md"},{"id":"breeze_tts","family":"breeze_tts","name":"BreezeTTS 2","task":"speech-synthesis","tasks":["tts","clone","design"],"summary":"T5Gemma2 encodes text and voice instructions. A Qwen3 temporal AR decoder and a Llama-style depth AR decoder generate audio codebooks. Cloning adds reference audio codes and their transcript, not a separate speaker-embedding network.","routes":[{"name":"Speech / voice design / cloning","nodes":[{"id":"text","kind":"input","label":"Text + optional voice instruction"},{"id":"reference","kind":"input","label":"Reference audio + transcript","optional":true},{"id":"tok","component":"frontend-bpe","detail":"Gemma tokenizer with speaker and instruction tags."},{"id":"textenc","component":"encoder-t5gemma2"},{"id":"refenc","component":"encoder-breeze-codec"},{"id":"ar","component":"ar-qwen3","detail":"Temporal AR: contextual text embeddings and optional reference codes form the prompt. Predicts the first codebook and a hidden state per frame; completed frame embeddings feed subsequent steps."},{"id":"depth","component":"ar-breeze-depth"},{"id":"codec","component":"codec-qwen3-tts-codec"},{"id":"out","kind":"output","label":"24 kHz speech"}],"edges":[{"from":"text","to":"tok"},{"from":"reference","to":"tok","label":"Reference transcript","optional":true},{"from":"tok","to":"textenc"},{"from":"textenc","to":"ar"},{"from":"reference","to":"refenc","label":"Reference waveform","optional":true},{"from":"refenc","to":"ar","label":"Audio prompt codes","optional":true},{"from":"ar","to":"depth","label":"Hidden state + first code"},{"from":"depth","to":"codec","label":"Complete audio frame"},{"from":"codec","to":"out"}]}],"sources":["https://huggingface.co/BreezeBlue/Breeze-TTS-2/blob/main/config.json"],"packages":[{"id":"breeze_tts_2_q8_0","display_name":"BreezeTTS 2 Q8_0 GGUF","precision":"q8_0"},{"id":"breeze_tts_2_bf16","display_name":"BreezeTTS 2 BF16 GGUF","precision":"bf16"}],"docs":["docs/tts.md","docs/gguf.md"],"variants":["BreezeTTS 2"],"inputs":[{"type":"text","label":"Text","required":true},{"type":"text","label":"Voice instruction","required":false},{"type":"audio","label":"Reference voice","required":false},{"type":"text","label":"Transcript when cloning","required":false}],"detailStatus":"Text conditioning, temporal/depth AR and reference-codec branches audited.","usageDoc":"docs/models/breeze_tts.md"},{"id":"cosyvoice3","family":"cosyvoice3","name":"CosyVoice3","task":"speech-synthesis","tasks":["tts","clone"],"summary":"Qwen2 generates S3 speech tokens, a conditional DiT turns them into mel features, and HiFT reconstructs speech. Reference audio has three branches: speech tokens, CAM++ speaker identity and mel features. Zero-shot mode additionally places reference tokens and transcript in the AR prompt.","routes":[{"name":"Zero-shot voice cloning","nodes":[{"id":"text","kind":"input","label":"Speech text + reference transcript"},{"id":"ref","kind":"input","label":"Reference voice"},{"id":"tok","component":"frontend-bpe"},{"id":"s3","component":"encoder-s3"},{"id":"speaker","component":"encoder-campplus"},{"id":"mel","component":"frontend-logmel"},{"id":"ar","component":"ar-qwen2"},{"id":"flow","component":"flow-cosyvoice3"},{"id":"codec","component":"codec-hift"},{"id":"out","kind":"output","label":"Speech waveform"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"ar"},{"from":"ref","to":"s3"},{"from":"ref","to":"speaker"},{"from":"ref","to":"mel"},{"from":"s3","to":"ar","label":"Reference speech prompt"},{"from":"s3","to":"flow","label":"Prompt tokens"},{"from":"speaker","to":"flow"},{"from":"mel","to":"flow"},{"from":"ar","to":"flow","label":"Generated speech tokens"},{"from":"flow","to":"codec"},{"from":"codec","to":"out"}]},{"name":"Cross-lingual / instruction-conditioned","inputs":[{"type":"text","label":"Speech text / instruction","required":true},{"type":"audio","label":"Reference voice","required":true}],"nodes":[{"id":"text","kind":"input","label":"Speech text / instruction"},{"id":"ref","kind":"input","label":"Reference voice"},{"id":"tok","component":"frontend-bpe"},{"id":"s3","component":"encoder-s3"},{"id":"speaker","component":"encoder-campplus"},{"id":"mel","component":"frontend-logmel"},{"id":"ar","component":"ar-qwen2","detail":"These templates omit reference speech tokens from the AR prompt."},{"id":"flow","component":"flow-cosyvoice3"},{"id":"codec","component":"codec-hift"},{"id":"out","kind":"output","label":"Speech waveform"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"ar"},{"from":"ref","to":"s3"},{"from":"ref","to":"speaker"},{"from":"ref","to":"mel"},{"from":"s3","to":"flow","label":"Prompt tokens"},{"from":"speaker","to":"flow"},{"from":"mel","to":"flow"},{"from":"ar","to":"flow","label":"Generated speech tokens"},{"from":"flow","to":"codec"},{"from":"codec","to":"out"}]}],"sources":["https://huggingface.co/FunAudioLLM/Fun-CosyVoice3-0.5B-2512/blob/main/cosyvoice3.yaml"],"packages":[{"id":"cosyvoice3_q8_0","display_name":"CosyVoice3 Q8_0 GGUF","precision":"q8_0"},{"id":"cosyvoice3_f32","display_name":"CosyVoice3 F32 GGUF","precision":"f32"}],"docs":["docs/tts.md","docs/gguf.md"],"variants":["CosyVoice3"],"inputs":[{"type":"text","label":"Speech text / optional instruction","required":true},{"type":"audio","label":"Reference voice","required":true},{"type":"text","label":"Reference transcript for zero-shot prompting","required":false}],"detailStatus":"Reference tokenizer, speaker/mel branches and template-specific AR conditioning audited.","usageDoc":"docs/models/cosyvoice3.md"},{"id":"vibevoice","family":"vibevoice","name":"VibeVoice","task":"speech-synthesis","tasks":["tts"],"summary":"Qwen2-family hidden states condition a residual-MLP diffusion head. A continuous acoustic codec reconstructs speech. Reference latents enter the AR prompt; generated speech also participates in the recurrent generation loop.","routes":[{"name":"Multi-speaker synthesis","nodes":[{"id":"text","kind":"input","label":"Speaker-tagged text"},{"id":"ref","kind":"input","label":"Reference voices","optional":true},{"id":"tok","component":"frontend-bpe"},{"id":"enc","component":"encoder-vibe-reference"},{"id":"ar","component":"ar-qwen2","detail":"Qwen2.5 family. Acoustic prompt latents use learned connectors."},{"id":"head","component":"flow-diffusion-head"},{"id":"codec","component":"codec-vibevoice-tokenizer"},{"id":"out","kind":"output","label":"24 kHz speech"}],"edges":[{"from":"text","to":"tok"},{"from":"ref","to":"enc","optional":true},{"from":"tok","to":"ar"},{"from":"enc","to":"ar","label":"Acoustic prompt","optional":true},{"from":"ar","to":"head","label":"Hidden conditioning"},{"from":"head","to":"codec","label":"Continuous latents"},{"from":"codec","to":"out"}]}],"sources":["https://huggingface.co/microsoft/VibeVoice-1.5B"],"packages":[{"id":"vibevoice_1_5b_q8_0","display_name":"VibeVoice 1.5B Q8_0 GGUF","precision":"q8_0"},{"id":"vibevoice_1_5b_bf16","display_name":"VibeVoice 1.5B BF16 GGUF","precision":"bf16"},{"id":"vibevoice_7b_q8_0","display_name":"VibeVoice 7B Q8_0 GGUF","precision":"q8_0"}],"docs":["docs/tts.md","docs/gguf.md"],"variants":["1.5B","7B"],"inputs":[{"type":"text","label":"Speaker-tagged script","required":true},{"type":"audio","label":"Reference voices","required":false}],"detailStatus":"Core conditioning path; recurrent acoustic feedback is summarized in text.","usageDoc":"docs/tts.md#vibevoice"},{"id":"dots_tts","family":"dots_tts","name":"DotTTS","task":"speech-synthesis","tasks":["tts","clone"],"summary":"Dots TTS combines Qwen2.5 autoregression with a patch-wise flow Transformer and a continuous AudioVAE. A causal semantic patch encoder feeds generated audio back into the language model. SOAR and MeanFlow differ in flow training and sampling, not by replacing the backbone with a discrete codec generator.","routes":[{"name":"TTS / speaker-only reference","inputs":[{"type":"text","label":"Target text","required":true},{"type":"audio","label":"Optional reference voice","required":false}],"nodes":[{"id":"text","kind":"input","label":"Target text"},{"id":"ref","kind":"input","label":"Optional reference voice","optional":true},{"id":"tok","component":"frontend-bpe"},{"id":"spk","component":"encoder-campplus"},{"id":"ar","component":"ar-qwen2","detail":"Qwen2.5-1.5B initialized backbone emits conditioning states and an EOS decision, not audio-code IDs."},{"id":"patch","component":"encoder-dots-patch"},{"id":"flow","component":"flow-soar-meanflow"},{"id":"vae","component":"codec-dots-audio-vae"},{"id":"out","kind":"output","label":"48 kHz speech"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"ar"},{"from":"ref","to":"spk","optional":true},{"from":"spk","to":"flow","optional":true},{"from":"patch","to":"ar","label":"Previous generated patch"},{"from":"ar","to":"flow"},{"from":"flow","to":"vae"},{"from":"vae","to":"out"}]},{"name":"Transcript-backed voice continuation","inputs":[{"type":"text","label":"Target + reference transcript","required":true},{"type":"audio","label":"Reference voice","required":true}],"nodes":[{"id":"text","kind":"input","label":"Target + reference text"},{"id":"ref","kind":"input","label":"Reference audio"},{"id":"tok","component":"frontend-bpe"},{"id":"spk","component":"encoder-campplus"},{"id":"enc","component":"encoder-dots-vae"},{"id":"patch","component":"encoder-dots-patch"},{"id":"ar","component":"ar-qwen2"},{"id":"flow","component":"flow-soar-meanflow"},{"id":"vae","component":"codec-dots-audio-vae"},{"id":"out","kind":"output","label":"Generated speech"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"ar"},{"from":"ref","to":"spk"},{"from":"ref","to":"enc"},{"from":"enc","to":"patch"},{"from":"patch","to":"ar","label":"Prompt / generated patch embeddings"},{"from":"spk","to":"flow"},{"from":"enc","to":"flow","label":"Prompt latent prefix"},{"from":"ar","to":"flow"},{"from":"flow","to":"vae"},{"from":"vae","to":"out"}]},{"name":"Edit checkpoint: source audio + instruction","inputs":[{"type":"audio","label":"Source speech","required":true},{"type":"text","label":"Edit instruction","required":true},{"type":"text","label":"Source / target text","required":false}],"nodes":[{"id":"source","kind":"input","label":"Source speech"},{"id":"text","kind":"input","label":"Instruction + optional transcripts"},{"id":"tok","component":"frontend-bpe","detail":"Edit tags and source/target text are arranged into the edit generation schedule."},{"id":"enc","component":"encoder-dots-vae"},{"id":"spk","component":"encoder-campplus","detail":"Source-speaker conditioning is controlled by the edit x-vector policy."},{"id":"patch","component":"encoder-dots-patch"},{"id":"ar","component":"ar-qwen2"},{"id":"flow","component":"flow-soar-meanflow"},{"id":"vae","component":"codec-dots-audio-vae"},{"id":"out","kind":"output","label":"Edited speech"}],"edges":[{"from":"source","to":"enc"},{"from":"source","to":"spk","optional":true},{"from":"text","to":"tok"},{"from":"tok","to":"ar"},{"from":"enc","to":"patch"},{"from":"patch","to":"ar"},{"from":"enc","to":"flow","label":"Source latent context"},{"from":"spk","to":"flow","optional":true},{"from":"ar","to":"flow"},{"from":"flow","to":"vae"},{"from":"vae","to":"out"}]}],"sources":["https://huggingface.co/dots-studio/dots.tts-soar","https://github.com/studio-dots-ai/dots.tts","https://huggingface.co/dots-studio/dots.tts.edit"],"packages":[{"id":"dots_tts_soar_q8_0","display_name":"DotTTS SOAR Q8_0 GGUF","precision":"q8_0"},{"id":"dots_tts_soar_bf16","display_name":"DotTTS SOAR BF16 GGUF","precision":"bf16"},{"id":"dots_tts_mf_q8_0","display_name":"DotTTS MeanFlow Q8_0 GGUF","precision":"q8_0"},{"id":"dots_tts_mf_bf16","display_name":"DotTTS MeanFlow BF16 GGUF","precision":"bf16"},{"id":"dots_tts_edit_q8_0","display_name":"DotTTS Edit Q8_0 GGUF","precision":"q8_0"},{"id":"dots_tts_edit_bf16","display_name":"DotTTS Edit BF16 GGUF","precision":"bf16"}],"docs":["docs/tts.md","docs/gguf.md"],"variants":["DotTTS SOAR","DotTTS MeanFlow","DotTTS Edit"],"inputs":[{"type":"text","label":"Target text or edit instruction","required":true},{"type":"audio","label":"Optional reference; required edit source","required":false}],"detailStatus":"Speaker-only conditioning, transcript-backed continuation and source-audio editing are separate routes. No discrete audio codebook is used. Generated patch feedback is described inside the patch encoder to keep the pipeline acyclic.","usageDoc":"docs/models/dots_tts.md"},{"id":"mira_tts","family":"mira_tts","name":"MiraTTS","task":"speech-synthesis","tasks":["tts","clone"],"summary":"MiraTTS is a Spark-TTS-derived system: a Qwen2-family AR generator predicts speech tokens, while ECAPA features and a Perceiver tokenize the reference voice. A speaker-conditioned ConvNeXt processor feeds a DAC-style waveform decoder, followed by FlashSR upsampling.","routes":[{"name":"Reference-voice TTS","nodes":[{"id":"text","kind":"input","label":"Target text"},{"id":"ref","kind":"input","label":"Reference voice"},{"id":"tok","component":"frontend-bpe","detail":"Qwen2 BPE with task, text, speaker-context and speech-generation markers."},{"id":"speaker","component":"encoder-mira-ecapa-perceiver"},{"id":"ar","component":"ar-qwen2"},{"id":"processor","component":"head-convnext"},{"id":"decoder","component":"codec-snake-conv-decoder"},{"id":"sr","component":"encoder-flashsr-residual-cnn","detail":"FlashSR is always applied after the 16 kHz decoder in this integration."},{"id":"out","kind":"output","label":"48 kHz speech"}],"edges":[{"from":"text","to":"tok"},{"from":"ref","to":"speaker"},{"from":"tok","to":"ar"},{"from":"speaker","to":"ar","label":"Context-token prefix"},{"from":"speaker","to":"processor","label":"Speaker conditioning"},{"from":"ar","to":"processor","label":"Speech-token IDs"},{"from":"processor","to":"decoder"},{"from":"decoder","to":"sr","label":"16 kHz waveform"},{"from":"sr","to":"out"}]}],"sources":["https://github.com/ysharma3501/MiraTTS","https://github.com/ysharma3501/FlashSR"],"packages":[],"docs":["docs/community_models/mira_tts.md"],"variants":["MiraTTS"],"inputs":[{"type":"text","label":"Target text","required":true},{"type":"audio","label":"Reference voice","required":true}],"detailStatus":"Required reference audio, 32 FSQ speaker-context tokens, acoustic conditioning and final FlashSR stage audited. No reference transcript is required; the current streaming path emits completed text segments.","usageDoc":"docs/community_models/mira_tts.md"},{"id":"maya1","family":"maya1","name":"Maya1","task":"speech-synthesis","tasks":["tts"],"summary":"Llama-architecture AR speech generation conditioned by a written voice description and emotion tags. Multi-scale SNAC tokens are decoded to speech; there is no reference-recording input.","routes":[{"name":"Voice design","nodes":[{"id":"text","kind":"input","label":"Text + voice description"},{"id":"tok","component":"frontend-bpe"},{"id":"ar","component":"ar-llama","detail":"Maya1 3B; instruction and text form the token prompt."},{"id":"codec","component":"codec-snac"},{"id":"out","kind":"output","label":"24 kHz speech"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"ar"},{"from":"ar","to":"codec","label":"Interleaved audio tokens"},{"from":"codec","to":"out"}]}],"sources":["https://huggingface.co/maya-research/maya1"],"packages":[{"id":"maya1_orig","display_name":"Maya1 Original-Precision GGUF","precision":"orig"},{"id":"maya1_q8_0","display_name":"Maya1 Q8_0 GGUF","precision":"q8_0"}],"docs":["docs/models/maya1.md","docs/gguf.md"],"variants":["3B"],"inputs":[{"type":"text","label":"Text","required":true},{"type":"text","label":"Voice description","required":true}],"detailStatus":"Text-only voice design, not reference-audio cloning.","usageDoc":"docs/models/maya1.md"},{"id":"outetts","family":"outetts","name":"OuteTTS","task":"speech-synthesis","tasks":["tts","clone"],"summary":"OuteTTS uses a Llama AR model to generate serialized speech codes and a DAC decoder for waveform synthesis. For cloning, audio.cpp first builds a word-aligned reference profile using DAC encoding, acoustic statistics and a companion Qwen3 Forced Aligner.","routes":[{"name":"Text to speech","inputs":[{"type":"text","label":"Target text","required":true}],"nodes":[{"id":"text","kind":"input","label":"Target text"},{"id":"tok","component":"frontend-bpe"},{"id":"ar","component":"ar-llama"},{"id":"codec","component":"codec-dac","detail":"Two-codebook OuteTTS audio codec."},{"id":"out","kind":"output","label":"Speech waveform"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"ar"},{"from":"ar","to":"codec"},{"from":"codec","to":"out"}]},{"name":"Reference voice cloning","inputs":[{"type":"text","label":"Target text","required":true},{"type":"audio","label":"Reference voice","required":true},{"type":"text","label":"Reference transcript","required":true}],"nodes":[{"id":"text","kind":"input","label":"Target text"},{"id":"ref","kind":"input","label":"Reference voice"},{"id":"transcript","kind":"input","label":"Reference transcript"},{"id":"encoder","component":"encoder-dac-reference"},{"id":"align","component":"encoder-oute-alignment"},{"id":"profile","component":"frontend-oute-profile"},{"id":"tok","component":"frontend-bpe"},{"id":"ar","component":"ar-llama"},{"id":"codec","component":"codec-dac","detail":"Two-codebook OuteTTS audio codec."},{"id":"out","kind":"output","label":"Cloned speech"}],"edges":[{"from":"ref","to":"encoder"},{"from":"ref","to":"align"},{"from":"transcript","to":"align"},{"from":"ref","to":"profile","label":"Acoustic statistics"},{"from":"encoder","to":"profile"},{"from":"align","to":"profile"},{"from":"transcript","to":"profile"},{"from":"text","to":"tok"},{"from":"profile","to":"tok"},{"from":"tok","to":"ar"},{"from":"ar","to":"codec"},{"from":"codec","to":"out"}]}],"sources":["https://huggingface.co/OuteAI/Llama-OuteTTS-1.0-1B"],"packages":[{"id":"outetts_1_0_1b_q8_0","display_name":"Llama-OuteTTS 1.0 1B Q8_0 GGUF","precision":"q8_0"}],"docs":["docs/community_models/outetts.md","docs/reports/outetts_validation.md","docs/gguf.md"],"variants":["Llama-OuteTTS 1.0 1B"],"inputs":[{"type":"text","label":"Target text","required":true},{"type":"audio","label":"Reference voice for cloning","required":false},{"type":"text","label":"Reference transcript for cloning","required":false}],"detailStatus":"Plain TTS and reference-cloning routes separated; the alignment companion is explicitly distinguished from OuteTTS's trained backbone.","usageDoc":"docs/community_models/outetts.md"},{"id":"glm_tts","family":"glm_tts","name":"GLM-TTS","task":"speech-synthesis","tasks":["tts","clone"],"summary":"Llama predicts discrete speech tokens from text and reference speech codes. A ConvNeXtV2-conditioned flow DiT generates mel features using reference mel and CAM++ identity, then HiFT synthesizes audio. The TTS and cloning tasks use the same reference-conditioned pipeline.","routes":[{"name":"Reference-conditioned TTS / cloning","nodes":[{"id":"text","kind":"input","label":"Speech text + reference transcript"},{"id":"ref","kind":"input","label":"Reference voice"},{"id":"tok","component":"frontend-bpe","detail":"ChatGLM text-token vocabulary for the Llama-based speech generator; tokenizer branding does not determine the backbone."},{"id":"vq","component":"encoder-glm-whisper-vq"},{"id":"speaker","component":"encoder-campplus"},{"id":"mel","component":"frontend-logmel"},{"id":"ar","component":"ar-llama"},{"id":"flow","component":"flow-flow-matching"},{"id":"codec","component":"codec-hift"},{"id":"out","kind":"output","label":"24 kHz speech"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"ar"},{"from":"ref","to":"vq"},{"from":"ref","to":"speaker"},{"from":"ref","to":"mel"},{"from":"vq","to":"ar","label":"Reference speech tokens"},{"from":"vq","to":"flow","label":"Prompt tokens"},{"from":"speaker","to":"flow","label":"Adaptive normalization condition"},{"from":"mel","to":"flow"},{"from":"ar","to":"flow","label":"Generated speech tokens"},{"from":"flow","to":"codec"},{"from":"codec","to":"out"}]}],"sources":["https://huggingface.co/zai-org/GLM-TTS"],"packages":[{"id":"glm_tts_q8_0","display_name":"GLM-TTS mixed Q8_0/F16 GGUF","precision":"q8_0"}],"docs":["docs/community_models/glm_tts.md","docs/reports/glm_tts_validation.md","docs/gguf.md"],"variants":["GLM-TTS"],"inputs":[{"type":"text","label":"Speech text + exact reference transcript","required":true},{"type":"audio","label":"Reference voice","required":true}],"detailStatus":"Reference-conditioned TTS/cloning route, Whisper-VQ and acoustic flow audited.","usageDoc":"docs/community_models/glm_tts.md"},{"id":"chatterbox","family":"chatterbox","name":"Chatterbox","task":"speech-synthesis","tasks":["tts","clone","vc"],"summary":"Llama-based T3 predicts S3 speech tokens from text and reference conditioning. S3Gen flow generates mel features and HiFT synthesizes speech. Voice conversion instead tokenizes the source recording and bypasses T3 entirely.","routes":[{"name":"English / multilingual voice cloning","nodes":[{"id":"text","kind":"input","label":"Text + language"},{"id":"ref","kind":"input","label":"Reference voice"},{"id":"tok","component":"frontend-bpe","detail":"Language-specific text vocabulary and normalization."},{"id":"speaker","component":"encoder-chatterbox-voice"},{"id":"s3","component":"encoder-chatterbox-s3"},{"id":"prompt","component":"encoder-chatterbox-prompt"},{"id":"camp","component":"encoder-campplus"},{"id":"mel","component":"frontend-logmel","detail":"S3Gen reference mel features; separate frontend settings from the speaker encoders."},{"id":"ar","component":"ar-llama","detail":"T3 uses projected speaker identity, resampled speech prompt and emotion conditioning. Learned text/speech position embeddings supplement the Llama-style backbone."},{"id":"flow","component":"flow-s3gen"},{"id":"vocoder","component":"codec-hift"},{"id":"out","kind":"output","label":"24 kHz speech"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"ar"},{"from":"ref","to":"speaker"},{"from":"speaker","to":"ar","label":"T3 speaker identity"},{"from":"ref","to":"s3"},{"from":"s3","to":"prompt"},{"from":"prompt","to":"ar"},{"from":"ref","to":"camp"},{"from":"camp","to":"flow","label":"Acoustic speaker identity"},{"from":"ref","to":"mel"},{"from":"mel","to":"flow","label":"Prompt mel"},{"from":"s3","to":"flow","label":"Prompt codes"},{"from":"ar","to":"flow","label":"Generated speech codes"},{"from":"flow","to":"vocoder"},{"from":"vocoder","to":"out"}]},{"name":"Voice conversion","inputs":[{"type":"audio","label":"Source speech","required":true},{"type":"audio","label":"Target reference voice","required":true}],"nodes":[{"id":"source","kind":"input","label":"Source speech"},{"id":"ref","kind":"input","label":"Target reference voice"},{"id":"sourcecodes","component":"encoder-chatterbox-s3"},{"id":"refcodes","component":"encoder-chatterbox-s3"},{"id":"camp","component":"encoder-campplus"},{"id":"mel","component":"frontend-logmel"},{"id":"flow","component":"flow-s3gen"},{"id":"vocoder","component":"codec-hift"},{"id":"out","kind":"output","label":"Converted speech"}],"edges":[{"from":"source","to":"sourcecodes"},{"from":"sourcecodes","to":"flow","label":"Source content tokens"},{"from":"ref","to":"refcodes"},{"from":"refcodes","to":"flow","label":"Prompt tokens"},{"from":"ref","to":"camp"},{"from":"camp","to":"flow","label":"Target identity"},{"from":"ref","to":"mel"},{"from":"mel","to":"flow"},{"from":"flow","to":"vocoder"},{"from":"vocoder","to":"out"}]}],"sources":["https://huggingface.co/ResembleAI/chatterbox"],"packages":[{"id":"chatterbox_q8_0","display_name":"Chatterbox Q8_0 GGUF","precision":"q8_0"},{"id":"chatterbox_f16","display_name":"Chatterbox F16 GGUF","precision":"f16"},{"id":"chatterbox_safetensors","display_name":"Chatterbox Safetensors","precision":"native"}],"docs":["docs/tts.md","docs/gguf.md"],"variants":["English","Multilingual T3 v2","Multilingual T3 v3"],"inputs":[{"type":"text","label":"Text","required":true},{"type":"audio","label":"Reference voice","required":true}],"detailStatus":"TTS reference branches and AR-free voice conversion audited against the current AudioCPP path.","usageDoc":"docs/tts.md#chatterbox"},{"id":"chatterbox_turbo","family":"chatterbox_turbo","name":"Chatterbox Turbo","task":"speech-synthesis","tasks":["tts"],"summary":"GPT-2 AR predicts speech tokens using a packaged voice prompt. A distilled S3Gen MeanFlow decoder produces mel features, then HiFT reconstructs the waveform. Unlike regular Chatterbox, this AudioCPP path does not accept custom reference audio.","routes":[{"name":"Packaged voice TTS","nodes":[{"id":"text","kind":"input","label":"Text"},{"id":"voice","kind":"input","label":"Packaged voice conditioning"},{"id":"tok","component":"frontend-bpe","detail":"GPT-2 BPE vocabulary."},{"id":"ar","component":"ar-gpt-2","detail":"Projected speaker vector and speech-prompt tokens precede text. No Perceiver or emotion conditioning is used in Turbo."},{"id":"flow","component":"flow-s3gen","detail":"Distilled MeanFlow: start and end times jointly condition the flow estimator."},{"id":"vocoder","component":"codec-hift"},{"id":"out","kind":"output","label":"24 kHz speech"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"ar"},{"from":"voice","to":"ar","label":"Speaker + prompt codes"},{"from":"voice","to":"flow","label":"Speaker / prompt codes / mel"},{"from":"ar","to":"flow","label":"Speech tokens"},{"from":"flow","to":"vocoder","label":"Mel features"},{"from":"vocoder","to":"out"}]}],"sources":["https://huggingface.co/ResembleAI/chatterbox-turbo"],"packages":[{"id":"chatterbox_turbo_q8_0","display_name":"Chatterbox Turbo Q8_0 GGUF","precision":"q8_0"},{"id":"chatterbox_turbo_f16","display_name":"Chatterbox Turbo F16 GGUF","precision":"f16"}],"docs":["docs/tts.md","docs/gguf.md"],"variants":["Chatterbox Turbo"],"inputs":[{"type":"text","label":"Text","required":true}],"detailStatus":"Current AudioCPP path uses packaged conditioning; custom reference-audio cloning is not exposed.","usageDoc":"docs/community_models/chatterbox_turbo.md"},{"id":"moss_tts_v15","family":"moss_tts_v15","name":"MOSS-TTS v1.5","task":"speech-synthesis","tasks":["tts","clone"],"summary":"Text and optional reference codes condition Qwen3 with delayed codebook generation. Reference audio is encoded by the MOSS tokenizer itself. The v1 causal audio decoder reconstructs the generated speech.","routes":[{"name":"TTS / reference voice conditioning","nodes":[{"id":"text","kind":"input","label":"Text + optional instruction"},{"id":"ref","kind":"input","label":"Reference audio","optional":true},{"id":"tok","component":"frontend-bpe"},{"id":"enc","component":"encoder-moss-reference","detail":"v1 codec reference encoding."},{"id":"ar","component":"ar-qwen3","detail":"8B delay-model backbone; text and reference-code embeddings share the prompt."},{"id":"heads","component":"head-delayed-codebooks"},{"id":"codec","component":"codec-moss-audio-tokenizer","detail":"v1; 24 kHz mono."},{"id":"out","kind":"output","label":"Speech waveform"}],"edges":[{"from":"text","to":"tok"},{"from":"ref","to":"enc","optional":true},{"from":"tok","to":"ar"},{"from":"enc","to":"ar","label":"Reference codes","optional":true},{"from":"ar","to":"heads"},{"from":"heads","to":"codec"},{"from":"codec","to":"out"}]}],"sources":["https://huggingface.co/OpenMOSS-Team/MOSS-TTS-v1.5"],"packages":[{"id":"moss_tts_v15_q8_0_codec_f16","display_name":"MOSS-TTS-v1.5 Q8_0 GGUF","precision":"q8_0"},{"id":"moss_tts_v15_q4_k_codec_f16","display_name":"MOSS-TTS-v1.5 Q4_K GGUF","precision":"q4_k"},{"id":"moss_tts_v15_bf16_codec_f16","display_name":"MOSS-TTS-v1.5 BF16 GGUF","precision":"bf16"}],"docs":["docs/community_models/moss_tts_v15.md","docs/tts.md","docs/gguf.md"],"variants":["MOSS-TTS-v1.5"],"inputs":[{"type":"text","label":"Speech text","required":true},{"type":"text","label":"Instruction / language","required":false},{"type":"audio","label":"Reference voice","required":false}],"detailStatus":"Optional reference-code conditioning and delayed generation audited.","usageDoc":"docs/community_models/moss_tts_v15.md"},{"id":"moss_ttsd","family":"moss_ttsd","name":"MOSS-TTSD","task":"speech-synthesis","tasks":["tts","clone"],"summary":"Speaker-tagged text and optional per-speaker reference codes prompt Qwen3 and delayed audio heads. The causal codec decodes prompt and generated audio together before the prompt portion is trimmed. Speaker tags select prompt identities, not separate speaker-classifier outputs.","routes":[{"name":"Multi-speaker dialogue","nodes":[{"id":"text","kind":"input","label":"Dialogue + optional reference transcript"},{"id":"ref","kind":"input","label":"Speaker reference recordings","optional":true},{"id":"tok","component":"frontend-bpe"},{"id":"enc","component":"encoder-moss-reference","detail":"v1; one reference-code sequence per supplied speaker."},{"id":"ar","component":"ar-qwen3","detail":"Dialogue prompt keeps positional speaker tags and reference spans."},{"id":"heads","component":"head-delayed-codebooks"},{"id":"codec","component":"codec-moss-audio-tokenizer","detail":"v1; causal prompt context retained, then prompt waveform trimmed."},{"id":"out","kind":"output","label":"Dialogue waveform"}],"edges":[{"from":"text","to":"tok"},{"from":"ref","to":"enc","optional":true},{"from":"tok","to":"ar"},{"from":"enc","to":"ar","label":"Speaker reference codes","optional":true},{"from":"ar","to":"heads"},{"from":"heads","to":"codec"},{"from":"codec","to":"out"}]}],"sources":["https://huggingface.co/OpenMOSS-Team/MOSS-TTSD-v1.0"],"packages":[{"id":"moss_ttsd_q8_0_codec_f16","display_name":"MOSS-TTSD Q8_0 GGUF","precision":"q8_0"},{"id":"moss_ttsd_q4_k_codec_f16","display_name":"MOSS-TTSD Q4_K GGUF (f16 heads)","precision":"q4_k"},{"id":"moss_ttsd_bf16_codec_f16","display_name":"MOSS-TTSD BF16 GGUF","precision":"bf16"}],"docs":["docs/community_models/moss_ttsd.md","docs/tts.md","docs/gguf.md"],"variants":["MOSS-TTSD"],"inputs":[{"type":"text","label":"Speaker-tagged dialogue","required":true},{"type":"audio","label":"Speaker reference recordings","required":false},{"type":"text","label":"Reference transcript / instruction","required":false}],"detailStatus":"Positional speaker-reference and dialogue continuation branches audited.","usageDoc":"docs/community_models/moss_ttsd.md"},{"id":"moss_voicegen","family":"moss_voicegen","name":"MOSS VoiceGenerator","task":"speech-synthesis","tasks":["design"],"summary":"Written speaker descriptions condition Qwen3-1.7B and delayed audio-codebook heads. MOSS Audio Tokenizer v1 decodes the generated codes. This family designs voices from text; it does not encode a reference recording.","routes":[{"name":"Text-conditioned voice design","nodes":[{"id":"text","kind":"input","label":"Text + voice description"},{"id":"tok","component":"frontend-bpe"},{"id":"ar","component":"ar-qwen3","detail":"Qwen3-1.7B; summed text/audio row embeddings."},{"id":"heads","component":"head-delayed-codebooks"},{"id":"codec","component":"codec-moss-audio-tokenizer","detail":"v1; 24 kHz mono, 16 generated codebooks."},{"id":"out","kind":"output","label":"Speech waveform"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"ar"},{"from":"ar","to":"heads"},{"from":"heads","to":"codec"},{"from":"codec","to":"out"}]}],"sources":["https://huggingface.co/OpenMOSS-Team/MOSS-VoiceGenerator"],"packages":[{"id":"moss_voicegen_bf16_codec_f16_decode","display_name":"MOSS-VoiceGenerator BF16 GGUF","precision":"bf16"}],"docs":["docs/community_models/moss_voicegen.md","docs/tts.md","docs/gguf.md"],"variants":["MOSS-VoiceGenerator"],"inputs":[{"type":"text","label":"Speech text","required":true},{"type":"text","label":"Voice description","required":false}],"detailStatus":"Voice-design input and decoder-only codec route audited.","usageDoc":"docs/community_models/moss_voicegen.md"},{"id":"moss_tts_local","family":"moss_tts_local","name":"MOSS-TTS Local","task":"speech-synthesis","tasks":["tts","clone"],"summary":"Qwen3 models the temporal sequence; a separate rotary depth Transformer predicts each frame's audio codebooks. Optional reference recordings use MOSS tokenizer encoding, while the v2 decoder produces stereo audio. This is a depth-decoder architecture, not delayed parallel heads.","routes":[{"name":"TTS / voice cloning","nodes":[{"id":"text","kind":"input","label":"Text + language"},{"id":"ref","kind":"input","label":"Reference voice","optional":true},{"id":"tok","component":"frontend-bpe"},{"id":"enc","component":"encoder-moss-reference","detail":"MOSS Audio Tokenizer v2 reference codes."},{"id":"global","component":"ar-qwen3"},{"id":"depth","component":"ar-local-transformer"},{"id":"codec","component":"codec-moss-audio-tokenizer","detail":"v2; 48 kHz stereo."},{"id":"out","kind":"output","label":"Stereo speech"}],"edges":[{"from":"text","to":"tok"},{"from":"ref","to":"enc","optional":true},{"from":"tok","to":"global"},{"from":"enc","to":"global","optional":true,"label":"Reference codes"},{"from":"global","to":"depth","label":"Frame state"},{"from":"depth","to":"codec","label":"Complete codebooks"},{"from":"codec","to":"out"}]}],"sources":["https://huggingface.co/OpenMOSS-Team/MOSS-TTS-Local-Transformer-v1.5"],"packages":[{"id":"moss_tts_local_v1_5_q8_0","display_name":"MOSS-TTS-Local v1.5 Q8_0 GGUF","precision":"q8_0"},{"id":"moss_tts_local_v1_5_bf16","display_name":"MOSS-TTS-Local v1.5 BF16 GGUF","precision":"bf16"}],"docs":["docs/models/moss_tts.md","docs/tts.md","docs/gguf.md"],"variants":["MOSS-TTS-Local v1.5"],"inputs":[{"type":"text","label":"Speech text / language","required":true},{"type":"audio","label":"Reference voice","required":false}],"detailStatus":"Qwen3 / depth decoder and reference-code conditioning audited.","usageDoc":"docs/models/moss_tts.md#moss-tts-local"},{"id":"moss_tts_nano","family":"moss_tts_nano","name":"MOSS-TTS Nano","task":"speech-synthesis","tasks":["tts","clone"],"summary":"Nano's custom global Transformer and local frame decoder generate text/control and audio codes. Optional reference audio is encoded with the Nano causal tokenizer. The same codec family reconstructs 48 kHz stereo speech, without a diffusion or mel-vocoder stage.","routes":[{"name":"TTS / reference continuation","nodes":[{"id":"text","kind":"input","label":"Text + optional reference transcript"},{"id":"ref","kind":"input","label":"Reference voice","optional":true},{"id":"tok","component":"frontend-bpe"},{"id":"enc","component":"encoder-moss-reference","detail":"Nano codec; optional transcript accompanies encoded reference audio."},{"id":"global","component":"ar-global-transformer"},{"id":"depth","component":"ar-moss-nano-depth"},{"id":"codec","component":"codec-moss-audio-tokenizer","detail":"Nano codec; 48 kHz stereo."},{"id":"out","kind":"output","label":"Stereo speech"}],"edges":[{"from":"text","to":"tok"},{"from":"ref","to":"enc","optional":true},{"from":"tok","to":"global"},{"from":"enc","to":"global","optional":true,"label":"Reference codes"},{"from":"global","to":"depth","label":"Frame state"},{"from":"depth","to":"codec","label":"Complete codebooks"},{"from":"codec","to":"out"}]}],"sources":["https://huggingface.co/OpenMOSS-Team/MOSS-TTS-Nano-100M"],"packages":[{"id":"moss_tts_nano_100m_q8_0","display_name":"MOSS-TTS-Nano 100M Q8_0 GGUF","precision":"q8_0"},{"id":"moss_tts_nano_100m_bf16","display_name":"MOSS-TTS-Nano 100M BF16 GGUF","precision":"bf16"}],"docs":["docs/models/moss_tts.md","docs/tts.md","docs/gguf.md"],"variants":["MOSS-TTS-Nano 100M"],"inputs":[{"type":"text","label":"Speech text","required":true},{"type":"audio","label":"Reference voice","required":false},{"type":"text","label":"Reference transcript","required":false}],"detailStatus":"Custom temporal/depth Transformers and Nano reference codec audited.","usageDoc":"docs/models/moss_tts.md#moss-tts-nano"},{"id":"fish_audio","family":"fish_audio","name":"Fish Audio S2 Pro","task":"speech-synthesis","tasks":["tts","clone"],"summary":"A slow AR Transformer generates semantic codes across time. Its hidden state and semantic code seed a fast AR Transformer that fills the remaining codebooks. Optional reference audio and transcripts condition the prompt; Fish DAC reconstructs the resulting frames.","routes":[{"name":"S2 Pro TTS / multi-reference cloning","nodes":[{"id":"text","kind":"input","label":"Text + optional reference transcripts"},{"id":"ref","kind":"input","label":"Reference recordings","optional":true},{"id":"tok","component":"frontend-bpe"},{"id":"enc","component":"encoder-fish-dac"},{"id":"slow","component":"ar-fish-slow"},{"id":"fast","component":"ar-fish-fast"},{"id":"codec","component":"codec-fish-dac"},{"id":"out","kind":"output","label":"Speech waveform"}],"edges":[{"from":"text","to":"tok"},{"from":"ref","to":"enc","optional":true},{"from":"tok","to":"slow"},{"from":"enc","to":"slow","optional":true,"label":"Reference codebooks"},{"from":"slow","to":"fast","label":"Hidden state + semantic code"},{"from":"fast","to":"codec","label":"Complete codebook frames"},{"from":"codec","to":"out"}]}],"sources":["https://huggingface.co/fishaudio/s2-pro"],"packages":[{"id":"fish_audio_s2_pro_q8_0","display_name":"Fish Audio S2 Pro Q8_0 GGUF","precision":"q8_0"},{"id":"fish_audio_s2_pro_bf16","display_name":"Fish Audio S2 Pro BF16 GGUF","precision":"bf16"}],"docs":["docs/tts.md","docs/gguf.md"],"variants":["Fish Audio S2 Pro"],"inputs":[{"type":"text","label":"Text / inline style tags","required":true},{"type":"audio","label":"One or more reference voices","required":false},{"type":"text","label":"Reference transcripts","required":false}],"detailStatus":"Slow/fast AR, reference-code prompting and codec reconstruction audited.","usageDoc":"docs/models/fish_audio.md"},{"id":"audio8_tts","family":"audio8_tts","name":"Audio8 TTS","task":"speech-synthesis","tasks":["tts","clone"],"summary":"Audio8 separates temporal semantic generation from within-frame codebook generation. Reference audio is encoded into prompt codes. The packaged 0.6B model uses a Qwen-style slow decoder; an alternate Falcon-H1 implementation is shown separately and is not a spec package.","routes":[{"name":"0.6B packaged TTS / cloning","nodes":[{"id":"text","kind":"input","label":"Text + optional reference transcripts"},{"id":"ref","kind":"input","label":"Reference recordings","optional":true},{"id":"tok","component":"frontend-bpe"},{"id":"enc","component":"encoder-audio8-codec"},{"id":"slow","component":"ar-audio8-slow"},{"id":"fast","component":"ar-audio8-fast"},{"id":"codec","component":"codec-transformer-convnext"},{"id":"out","kind":"output","label":"44.1 kHz speech"}],"edges":[{"from":"text","to":"tok"},{"from":"ref","to":"enc","optional":true},{"from":"tok","to":"slow"},{"from":"enc","to":"slow","optional":true,"label":"Reference codes"},{"from":"slow","to":"fast","label":"Semantic code + hidden state"},{"from":"fast","to":"codec"},{"from":"codec","to":"out"}]},{"name":"0.1B Falcon-H1 alternate code path","nodes":[{"id":"text","kind":"input","label":"Text + optional reference transcripts"},{"id":"ref","kind":"input","label":"Reference recordings","optional":true},{"id":"tok","component":"frontend-bpe"},{"id":"enc","component":"encoder-audio8-codec"},{"id":"slow","component":"ar-audio8-falcon"},{"id":"fast","component":"ar-audio8-fast","detail":"Compact alternate-backbone configuration; not the packaged 0.6B dimensions."},{"id":"codec","component":"codec-transformer-convnext"},{"id":"out","kind":"output","label":"Speech waveform"}],"edges":[{"from":"text","to":"tok"},{"from":"ref","to":"enc","optional":true},{"from":"tok","to":"slow"},{"from":"enc","to":"slow","optional":true,"label":"Reference codes"},{"from":"slow","to":"fast","label":"Semantic code + hidden state"},{"from":"fast","to":"codec"},{"from":"codec","to":"out"}]}],"sources":["https://huggingface.co/Edge0/Audio8-TTS-Preview-0.6b"],"packages":[{"id":"audio8_tts_preview_0_6b_q8_0","display_name":"Audio8 TTS Preview 0.6B Q8_0 GGUF","precision":"q8_0"}],"docs":[],"variants":["Audio8 TTS Preview 0.6B"],"inputs":[{"type":"text","label":"Speech text","required":true},{"type":"audio","label":"Reference recordings","required":false},{"type":"text","label":"Reference transcripts","required":false}],"detailStatus":"Packaged 0.6B route and alternate Falcon-H1 code architecture audited; alternate availability is not a synthesis validation claim.","usageDoc":"docs/community_models/audio8_tts.md"},{"id":"pocket_tts","family":"pocket_tts","name":"PocketTTS","task":"speech-synthesis","tasks":["tts","clone"],"summary":"PocketTTS embeds SentencePiece text and a reference-voice prefix into FlowLM. Its causal Transformer advances one continuous acoustic latent at a time; a small time-conditioned flow MLP generates each latent. Mimi decodes the latent stream to audio without discrete codebook lookup.","routes":[{"name":"Reference-audio voice cloning","inputs":[{"type":"text","label":"Target text","required":true},{"type":"audio","label":"Reference voice","required":true}],"nodes":[{"id":"text","kind":"input","label":"Target text"},{"id":"ref","kind":"input","label":"Reference voice"},{"id":"tokens","component":"frontend-pocket-text"},{"id":"encoder","component":"encoder-pocket-mimi"},{"id":"ar","component":"ar-flowlm"},{"id":"flow","component":"flow-flow-matching-head"},{"id":"decoder","component":"codec-pocket-mimi"},{"id":"out","kind":"output","label":"Streaming speech waveform"}],"edges":[{"from":"text","to":"tokens"},{"from":"tokens","to":"ar"},{"from":"ref","to":"encoder"},{"from":"encoder","to":"ar","label":"Voice prefix"},{"from":"ar","to":"flow"},{"from":"flow","to":"decoder","label":"Continuous latents"},{"from":"decoder","to":"out"}]},{"name":"Saved / packaged voice state","inputs":[{"type":"text","label":"Target text","required":true},{"type":"state","label":"Saved voice state","required":true}],"nodes":[{"id":"text","kind":"input","label":"Target text"},{"id":"state","kind":"input","label":"Saved voice prefix state"},{"id":"tokens","component":"frontend-pocket-text"},{"id":"ar","component":"ar-flowlm"},{"id":"flow","component":"flow-flow-matching-head"},{"id":"decoder","component":"codec-pocket-mimi"},{"id":"out","kind":"output","label":"Streaming speech waveform"}],"edges":[{"from":"text","to":"tokens"},{"from":"tokens","to":"ar"},{"from":"state","to":"ar"},{"from":"ar","to":"flow"},{"from":"flow","to":"decoder","label":"Continuous latents"},{"from":"decoder","to":"out"}]}],"sources":["https://github.com/kyutai-labs/pocket-tts"],"packages":[{"id":"pocket_tts_english_q8_0","display_name":"PocketTTS English Q8_0 GGUF","precision":"q8_0"},{"id":"pocket_tts_english_bf16","display_name":"PocketTTS English BF16 GGUF","precision":"bf16"},{"id":"pocket_tts_german_q8_0","display_name":"PocketTTS German Q8_0 GGUF","precision":"q8_0"},{"id":"pocket_tts_german_bf16","display_name":"PocketTTS German BF16 GGUF","precision":"bf16"},{"id":"pocket_tts_italian_q8_0","display_name":"PocketTTS Italian Q8_0 GGUF","precision":"q8_0"},{"id":"pocket_tts_italian_bf16","display_name":"PocketTTS Italian BF16 GGUF","precision":"bf16"},{"id":"pocket_tts_portuguese_q8_0","display_name":"PocketTTS Portuguese Q8_0 GGUF","precision":"q8_0"},{"id":"pocket_tts_portuguese_bf16","display_name":"PocketTTS Portuguese BF16 GGUF","precision":"bf16"},{"id":"pocket_tts_spanish_q8_0","display_name":"PocketTTS Spanish Q8_0 GGUF","precision":"q8_0"},{"id":"pocket_tts_spanish_bf16","display_name":"PocketTTS Spanish BF16 GGUF","precision":"bf16"},{"id":"pocket_tts_english_safetensors","display_name":"PocketTTS English Safetensors","precision":"native"}],"docs":["docs/tts.md","docs/gguf.md"],"variants":["PocketTTS English","PocketTTS German","PocketTTS Italian","PocketTTS Portuguese","PocketTTS Spanish"],"inputs":[{"type":"text","label":"Target text","required":true},{"type":"audio/state","label":"Reference voice or saved voice state","required":true}],"detailStatus":"Text, continuous voice prompt, per-frame flow head and stateful Mimi decoding audited. Saved voice states bypass reference encoding.","usageDoc":"docs/tts.md#pockettts"},{"id":"voxcpm1","family":"voxcpm1","name":"VoxCPM1","task":"speech-synthesis","tasks":["tts","clone"],"summary":"VoxCPM1 combines hierarchical MiniCPM autoregression with local diffusion over continuous AudioVAE patches. Optional prompt audio and its transcript provide voice and speaking-style context.","routes":[{"name":"TTS / prompt voice cloning","nodes":[{"id":"text","kind":"input","label":"Target + optional prompt text"},{"id":"audio","kind":"input","label":"Optional prompt audio","optional":true},{"id":"tok","component":"frontend-bpe"},{"id":"enc","component":"encoder-voxcpm-audiovae"},{"id":"local","component":"encoder-voxcpm-local"},{"id":"ar","component":"ar-voxcpm-hierarchy","detail":"V1 fuses semantic and local patch embeddings by addition."},{"id":"flow","component":"flow-voxcpm1-local"},{"id":"codec","component":"codec-voxcpm-audiovae","detail":"AudioVAE V1 checkpoint."},{"id":"out","kind":"output","label":"Speech waveform"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"ar"},{"from":"audio","to":"enc","optional":true},{"from":"enc","to":"local","optional":true},{"from":"local","to":"ar","optional":true},{"from":"ar","to":"flow"},{"from":"flow","to":"codec","label":"Continuous patches"},{"from":"codec","to":"out"}]}],"sources":["https://huggingface.co/openbmb/VoxCPM-0.5B"],"packages":[{"id":"voxcpm1_0_5b_q8_0","display_name":"VoxCPM 0.5B Q8_0 GGUF","precision":"q8_0"}],"docs":["docs/tts.md","docs/gguf.md"],"variants":["VoxCPM 0.5B"],"inputs":[{"type":"text","label":"Target text","required":true},{"type":"audio","label":"Optional voice prompt","required":false},{"type":"text","label":"Prompt transcript for cloning","required":false}],"detailStatus":"Reference encoding, hierarchical AR, semantic scalar bottleneck, V1 local diffusion and continuous AudioVAE synthesis audited.","usageDoc":"docs/community_models/voxcpm1.md"},{"id":"voxcpm2","family":"voxcpm2","name":"VoxCPM2","task":"speech-synthesis","tasks":["tts","clone","design"],"summary":"VoxCPM2 uses hierarchical MiniCPM-4 autoregression and patch-local diffusion. Reference-only voice conditioning does not require a transcript; continuation audio uses its transcript. AudioVAE V2 encodes 16 kHz references and decodes 48 kHz speech.","routes":[{"name":"TTS / voice reference / continuation","nodes":[{"id":"text","kind":"input","label":"Target text + optional continuation text"},{"id":"ref","kind":"input","label":"Optional reference voice","optional":true},{"id":"prompt","kind":"input","label":"Optional continuation audio","optional":true},{"id":"tok","component":"frontend-bpe"},{"id":"enc","component":"encoder-voxcpm-audiovae","detail":"Both audio prompts use AudioVAE V2 at 16 kHz; their roles in prompt assembly differ."},{"id":"local","component":"encoder-voxcpm-local"},{"id":"ar","component":"ar-voxcpm-hierarchy","detail":"V2 concatenates and projects semantic and acoustic embeddings for the residual LM. Reference voice precedes text; continuation audio follows it."},{"id":"flow","component":"flow-voxcpm2-local"},{"id":"codec","component":"codec-voxcpm-audiovae","detail":"AudioVAE V2 synthesizes 48 kHz audio."},{"id":"out","kind":"output","label":"48 kHz speech"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"ar"},{"from":"ref","to":"enc","optional":true},{"from":"prompt","to":"enc","optional":true},{"from":"enc","to":"local","optional":true},{"from":"local","to":"ar","optional":true},{"from":"ar","to":"flow"},{"from":"flow","to":"codec","label":"Continuous patches"},{"from":"codec","to":"out"}]}],"sources":["https://huggingface.co/openbmb/VoxCPM2"],"packages":[{"id":"voxcpm2_q8_0","display_name":"VoxCPM2 Q8_0 GGUF","precision":"q8_0"},{"id":"voxcpm2_bf16","display_name":"VoxCPM2 BF16 GGUF","precision":"bf16"},{"id":"voxcpm2_orig","display_name":"VoxCPM2 Original-Dtype GGUF","precision":"orig"},{"id":"voxcpm2_safetensors","display_name":"VoxCPM2 Safetensors","precision":"native"}],"docs":["docs/tts.md","docs/gguf.md"],"variants":["VoxCPM2"],"inputs":[{"type":"text","label":"Target text / inline voice instructions","required":true},{"type":"audio","label":"Optional reference voice","required":false},{"type":"audio","label":"Optional continuation prompt","required":false},{"type":"text","label":"Continuation transcript","required":false}],"detailStatus":"Reference-only and transcribed continuation inputs distinguished; V2 token-prefix diffusion conditioning and AudioVAE rate asymmetry audited.","usageDoc":"docs/tts.md#voxcpm2"},{"id":"sopro_tts","family":"sopro_tts","name":"Sopro V2 Turbo","task":"speech-synthesis","tasks":["tts","clone"],"summary":"Sopro V2 Turbo predicts semantic tokens autoregressively, then uses a conditional acoustic DiT to generate mel features for a Vocos-style decoder. Reference audio supplies semantic style tokens, speaker/style vectors and a mel prompt; no reference transcript is required.","routes":[{"name":"Zero-shot voice cloning","nodes":[{"id":"text","kind":"input","label":"Target text + language"},{"id":"ref","kind":"input","label":"Reference voice"},{"id":"tok","component":"frontend-sopro-text"},{"id":"sem","component":"encoder-sopro-semantic"},{"id":"speaker","component":"encoder-sopro-speaker"},{"id":"mel","component":"frontend-logmel","detail":"Vocoder mel frontend supplies the acoustic prompt; separate encoder frontend settings are inside their blocks."},{"id":"ar","component":"ar-semantic-transformer"},{"id":"flow","component":"flow-acoustic-dit"},{"id":"vocoder","component":"codec-vocos"},{"id":"out","kind":"output","label":"24 kHz speech"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"ar"},{"from":"ref","to":"sem"},{"from":"ref","to":"speaker"},{"from":"ref","to":"mel"},{"from":"sem","to":"ar","label":"Style + semantic prefix"},{"from":"sem","to":"flow","label":"Reference tokens"},{"from":"ar","to":"flow","label":"Generated tokens"},{"from":"speaker","to":"flow"},{"from":"mel","to":"flow","label":"Prompt mel"},{"from":"flow","to":"vocoder"},{"from":"vocoder","to":"out"}]}],"sources":["https://github.com/samuel-vitorino/sopro"],"packages":[{"id":"sopro_v2_turbo_f16","display_name":"Sopro V2 Turbo F16 GGUF","precision":"f16"},{"id":"sopro_v2_turbo_safetensors","display_name":"Sopro V2 Turbo (upstream safetensors)","precision":"orig"}],"docs":["docs/community_models/sopro_tts.md"],"variants":["V2 Turbo"],"inputs":[{"type":"text","label":"Target text / language","required":true},{"type":"audio","label":"Reference speaker audio","required":true}],"detailStatus":"Text frontend, both reference encoders, style pooling, semantic AR, acoustic DiT and Vocos path audited. Distinct from Soprano TTS; audio.cpp emits complete text segments rather than upstream's frame-streaming route.","usageDoc":"docs/community_models/sopro_tts.md"},{"id":"f5_tts","family":"f5_tts","name":"F5-TTS","task":"speech-synthesis","tasks":["tts","clone"],"summary":"F5-TTS combines ConvNeXt V2 text conditioning with a rotary DiT flow model that extends a reference mel prompt. Vocos turns the generated mel continuation into speech. The packaged Habibi variants specialize the F5 architecture for Arabic dialects.","routes":[{"name":"Reference-conditioned TTS / Habibi","nodes":[{"id":"text","kind":"input","label":"Target + reference transcript"},{"id":"ref","kind":"input","label":"Reference voice"},{"id":"tok","component":"frontend-f5-text"},{"id":"encoder","component":"encoder-text-convnext"},{"id":"mel","component":"frontend-logmel"},{"id":"flow","component":"flow-f5-dit"},{"id":"vocoder","component":"codec-vocos"},{"id":"out","kind":"output","label":"24 kHz speech"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"encoder"},{"from":"ref","to":"mel"},{"from":"encoder","to":"flow"},{"from":"mel","to":"flow","label":"Reference prompt"},{"from":"flow","to":"vocoder"},{"from":"vocoder","to":"out"}]}],"sources":["https://github.com/SWivid/F5-TTS","https://github.com/SWivid/Habibi-TTS"],"packages":[{"id":"habibi_unified","display_name":"Habibi-TTS Unified (Arabic, multi-dialect)","precision":"orig"},{"id":"vocos_mel_24khz","display_name":"Vocos mel 24kHz vocoder (GGUF)","precision":"orig"},{"id":"habibi_alg","display_name":"Habibi-TTS ALG specialized checkpoint (GGUF)","precision":"orig"},{"id":"habibi_egy","display_name":"Habibi-TTS EGY specialized checkpoint (GGUF)","precision":"orig"},{"id":"habibi_irq","display_name":"Habibi-TTS IRQ specialized checkpoint (GGUF)","precision":"orig"},{"id":"habibi_mar","display_name":"Habibi-TTS MAR specialized checkpoint (GGUF)","precision":"orig"},{"id":"habibi_msa","display_name":"Habibi-TTS MSA specialized checkpoint (GGUF)","precision":"orig"},{"id":"habibi_sau","display_name":"Habibi-TTS SAU specialized checkpoint (GGUF)","precision":"orig"},{"id":"habibi_uae","display_name":"Habibi-TTS UAE specialized checkpoint (GGUF)","precision":"orig"}],"docs":["docs/community_models/f5_tts.md"],"variants":["Habibi Unified","Habibi ALG","Habibi EGY","Habibi IRQ","Habibi MAR","Habibi MSA","Habibi SAU","Habibi UAE"],"inputs":[{"type":"text","label":"Target text","required":true},{"type":"audio","label":"Reference voice","required":true},{"type":"text","label":"Reference transcript","required":true},{"type":"control","label":"Habibi dialect","required":false}],"detailStatus":"Reference transcript, reference mel and text conditioning audited; this path is non-autoregressive.","usageDoc":"docs/community_models/f5_tts.md"},{"id":"zipvoice","family":"zipvoice","name":"ZipVoice","task":"speech-synthesis","tasks":["tts","clone"],"summary":"ZipVoice encodes reference and target text with a Zipformer and expands those features to acoustic frames. A second, time-conditioned multi-rate Zipformer generates mel features through flow matching, using the reference mel prompt. Vocos synthesizes the continuation.","routes":[{"name":"Zero-shot TTS","nodes":[{"id":"text","kind":"input","label":"Target + reference transcript"},{"id":"ref","kind":"input","label":"Reference voice"},{"id":"tok","component":"frontend-zipvoice-text"},{"id":"encoder","component":"encoder-zipvoice-text"},{"id":"mel","component":"frontend-logmel"},{"id":"flow","component":"flow-zipformer"},{"id":"vocoder","component":"codec-vocos"},{"id":"out","kind":"output","label":"24 kHz speech"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"encoder"},{"from":"ref","to":"mel"},{"from":"mel","to":"encoder","label":"Reference duration"},{"from":"encoder","to":"flow"},{"from":"mel","to":"flow","label":"Reference prompt"},{"from":"flow","to":"vocoder"},{"from":"vocoder","to":"out"}]}],"sources":["https://github.com/k2-fsa/ZipVoice"],"packages":[{"id":"zipvoice_distill_gguf","display_name":"ZipVoice-Distill GGUF (local conversion)","precision":"orig"},{"id":"zipvoice_distill_q8_0","display_name":"ZipVoice-Distill GGUF (Q8_0)","precision":"q8_0"}],"docs":["docs/community_models/zipvoice.md"],"variants":["ZipVoice-Distill"],"inputs":[{"type":"text","label":"Target text","required":true},{"type":"audio","label":"Reference voice","required":true},{"type":"text","label":"Reference transcript","required":true}],"detailStatus":"Text, reference transcript, duration expansion and acoustic prompt branches audited.","usageDoc":"docs/community_models/zipvoice.md"},{"id":"echo_tts","family":"echo_tts","name":"Echo-TTS","task":"speech-synthesis","tasks":["clone"],"summary":"Echo-TTS generates continuous acoustic latents with a joint-attention diffusion Transformer. A byte-level text encoder and a causal reference-latent encoder provide separate conditioning. PCA connects the diffusion representation to its Fish DAC audio codec.","routes":[{"name":"Reference-conditioned TTS","nodes":[{"id":"text","kind":"input","label":"Target text"},{"id":"ref","kind":"input","label":"Reference voice"},{"id":"textenc","component":"encoder-echo-text"},{"id":"codecenc","component":"encoder-fish-dac"},{"id":"pca","component":"dsp-echo-pca","detail":"Forward: codec latent to scaled PCA space."},{"id":"speaker","component":"encoder-echo-reference"},{"id":"flow","component":"flow-echo-dit"},{"id":"inverse","component":"dsp-echo-pca","detail":"Inverse: generated PCA latent to codec latent after output-length trimming."},{"id":"decode","component":"codec-echo-fish-dac"},{"id":"out","kind":"output","label":"Speech waveform"}],"edges":[{"from":"text","to":"textenc"},{"from":"ref","to":"codecenc"},{"from":"codecenc","to":"pca"},{"from":"pca","to":"speaker"},{"from":"textenc","to":"flow"},{"from":"speaker","to":"flow"},{"from":"flow","to":"inverse"},{"from":"inverse","to":"decode"},{"from":"decode","to":"out"}]}],"sources":["https://github.com/jordandare/echo-tts"],"packages":[{"id":"echo_tts_q8_0","display_name":"Echo-TTS Q8_0 GGUF","precision":"q8_0"},{"id":"echo_tts_f16","display_name":"Echo-TTS F16 GGUF","precision":"f16"}],"docs":[],"variants":["Echo-TTS"],"inputs":[{"type":"text","label":"Target text","required":true},{"type":"audio","label":"Reference speaker audio","required":true}],"detailStatus":"Byte frontend, required reference audio, causal speaker encoder, joint-attention diffusion and PCA/codec boundary audited. No reference transcript is required.","usageDoc":"docs/community_models/echo_tts.md"},{"id":"omnivoice","family":"omnivoice","name":"OmniVoice","task":"speech-synthesis","tasks":["tts","clone","design"],"summary":"OmniVoice uses Qwen3 blocks for discrete masked diffusion, not autoregressive generation. It predicts a fixed-length speech-code sequence conditioned on text, optional voice instructions or reference speech codes. A Higgs Audio V2 decoder synthesizes the waveform.","routes":[{"name":"Auto voice / voice design","inputs":[{"type":"text","label":"Target text","required":true},{"type":"text","label":"Voice description / language","required":false}],"nodes":[{"id":"text","kind":"input","label":"Text + optional voice description"},{"id":"tokens","component":"frontend-bpe"},{"id":"duration","component":"head-omnivoice-duration"},{"id":"diffusion","component":"flow-qwen3-diffusion"},{"id":"codec","component":"codec-higgs-audio-v2"},{"id":"out","kind":"output","label":"Speech waveform"}],"edges":[{"from":"text","to":"tokens"},{"from":"text","to":"duration"},{"from":"tokens","to":"diffusion"},{"from":"duration","to":"diffusion","label":"Target length"},{"from":"diffusion","to":"codec"},{"from":"codec","to":"out"}]},{"name":"Reference voice cloning","inputs":[{"type":"text","label":"Target text","required":true},{"type":"audio","label":"Reference voice","required":true},{"type":"text","label":"Reference transcript","required":false}],"nodes":[{"id":"text","kind":"input","label":"Target + optional reference text"},{"id":"ref","kind":"input","label":"Reference voice"},{"id":"tokens","component":"frontend-bpe"},{"id":"encoder","component":"encoder-higgs-v2-reference"},{"id":"duration","component":"head-omnivoice-duration"},{"id":"diffusion","component":"flow-qwen3-diffusion"},{"id":"codec","component":"codec-higgs-audio-v2"},{"id":"out","kind":"output","label":"Speech waveform"}],"edges":[{"from":"text","to":"tokens"},{"from":"text","to":"duration"},{"from":"ref","to":"encoder"},{"from":"ref","to":"duration","label":"Reference duration"},{"from":"tokens","to":"diffusion"},{"from":"encoder","to":"diffusion","label":"Reference codes"},{"from":"duration","to":"diffusion","label":"Target length"},{"from":"diffusion","to":"codec"},{"from":"codec","to":"out"}]}],"sources":["https://github.com/k2-fsa/OmniVoice/blob/main/omnivoice/models/omnivoice.py"],"packages":[{"id":"omnivoice_q8_0","display_name":"OmniVoice Q8_0 GGUF","precision":"q8_0"},{"id":"omnivoice_bf16","display_name":"OmniVoice BF16 GGUF","precision":"bf16"},{"id":"omnivoice_f16","display_name":"OmniVoice F16 GGUF","precision":"f16"},{"id":"omnivoice_safetensors","display_name":"OmniVoice Safetensors","precision":"native"}],"docs":["docs/models/omnivoice.md","docs/tts.md","docs/gguf.md"],"variants":["OmniVoice"],"inputs":[{"type":"text","label":"Target text","required":true},{"type":"text","label":"Voice design / language","required":false},{"type":"audio","label":"Reference voice","required":false},{"type":"text","label":"Reference transcript","required":false}],"detailStatus":"Auto voice, design and reference-cloning inputs audited. Duration is rule-based; audio.cpp streaming is chunked output, not native incremental model decoding.","usageDoc":"docs/models/omnivoice.md"},{"id":"irodori_tts","family":"irodori_tts","name":"Irodori-TTS","task":"speech-synthesis","tasks":["tts","clone","design"],"summary":"Text and optional voice instructions condition a rectified-flow DiT. Reference audio supplies DAC-VAE latent speaker states, while a learned duration predictor sets output length. v4 uses ModernBERT-JA; v3 uses custom rotary text/caption encoders.","routes":[{"name":"v4 / v4.1 Small / Anime","nodes":[{"id":"text","kind":"input","label":"Japanese text + optional instruction"},{"id":"ref","kind":"input","label":"Reference voice","optional":true},{"id":"textenc","component":"encoder-modernbert"},{"id":"refenc","component":"encoder-dac-vae"},{"id":"speaker","component":"encoder-irodori-speaker"},{"id":"duration","component":"head-irodori-duration"},{"id":"flow","component":"flow-rf-dit"},{"id":"codec","component":"codec-dac-vae"},{"id":"out","kind":"output","label":"Speech waveform"}],"edges":[{"from":"text","to":"textenc"},{"from":"ref","to":"refenc","optional":true},{"from":"refenc","to":"speaker","optional":true},{"from":"textenc","to":"duration"},{"from":"speaker","to":"duration","optional":true},{"from":"textenc","to":"flow"},{"from":"speaker","to":"flow","optional":true},{"from":"duration","to":"flow","label":"Predicted length unless overridden"},{"from":"flow","to":"codec"},{"from":"codec","to":"out"}]},{"name":"v3 / v3 VoiceDesign","nodes":[{"id":"text","kind":"input","label":"Japanese text / optional VoiceDesign caption"},{"id":"ref","kind":"input","label":"Reference voice","optional":true},{"id":"textenc","component":"encoder-text-caption-encoders"},{"id":"refenc","component":"encoder-dac-vae"},{"id":"speaker","component":"encoder-irodori-speaker"},{"id":"duration","component":"head-irodori-duration"},{"id":"flow","component":"flow-rf-dit"},{"id":"codec","component":"codec-dac-vae"},{"id":"out","kind":"output","label":"Speech waveform"}],"edges":[{"from":"text","to":"textenc"},{"from":"ref","to":"refenc","optional":true},{"from":"refenc","to":"speaker","optional":true},{"from":"textenc","to":"duration"},{"from":"speaker","to":"duration","optional":true},{"from":"textenc","to":"flow"},{"from":"speaker","to":"flow","optional":true},{"from":"duration","to":"flow","label":"Predicted length unless overridden"},{"from":"flow","to":"codec"},{"from":"codec","to":"out"}]}],"sources":["https://huggingface.co/Aratako/Irodori-TTS-500M-v3","https://huggingface.co/Aratako/Irodori-TTS-v4.1-Small","https://huggingface.co/Aratako/Irodori-TTS-600M-v3-VoiceDesign"],"packages":[{"id":"irodori_tts_v4_small_q8_0","display_name":"Irodori-TTS v4.1 Small Q8_0 GGUF","precision":"q8_0"},{"id":"irodori_tts_v4_1_anime_q8_0","display_name":"Irodori-TTS v4.1 Anime Q8_0 GGUF","precision":"q8_0"},{"id":"irodori_tts_v4_small_f16","display_name":"Irodori-TTS v4.1 Small F16 GGUF","precision":"f16"},{"id":"irodori_tts_600m_v3_voicedesign_q8_0","display_name":"Irodori-TTS 600M v3 VoiceDesign Q8_0 GGUF","precision":"q8_0"},{"id":"irodori_tts_600m_v3_voicedesign_f16","display_name":"Irodori-TTS 600M v3 VoiceDesign F16 GGUF","precision":"f16"},{"id":"irodori_tts_500m_v3_q8_0","display_name":"Irodori-TTS 500M v3 Q8_0 GGUF","precision":"q8_0"},{"id":"irodori_tts_500m_v3_f16","display_name":"Irodori-TTS 500M v3 F16 GGUF","precision":"f16"}],"docs":["docs/tts.md","docs/gguf.md"],"variants":["Irodori-TTS v4.1 Small","Irodori-TTS v4.1 Anime","Irodori-TTS 600M v3 VoiceDesign","Irodori-TTS 500M v3"],"inputs":[{"type":"text","label":"Japanese text","required":true},{"type":"audio","label":"Reference voice","required":false},{"type":"text","label":"Voice instruction on supported variants","required":false}],"detailStatus":"Variant-specific text encoders, continuous reference conditioning and learned duration path audited.","usageDoc":"docs/models/irodori_tts.md"},{"id":"dramabox","family":"dramabox","name":"DramaBox","task":"speech-synthesis","tasks":["tts","clone"],"summary":"DramaBox uses Gemma3 for prompt features, an LTX-derived audio diffusion Transformer for latent generation, a mel AudioVAE and two BigVGAN stages for waveform synthesis and bandwidth extension. Gemma3 is not the audio-token generator.","routes":[{"name":"Prompt-driven speech / optional voice reference","nodes":[{"id":"text","kind":"input","label":"Dialogue + delivery prompt"},{"id":"ref","kind":"input","label":"Optional voice reference","optional":true},{"id":"gemma","component":"encoder-gemma3"},{"id":"connector","component":"encoder-dramabox-connector"},{"id":"mel","component":"frontend-logmel"},{"id":"enc","component":"encoder-dramabox-vae"},{"id":"flow","component":"flow-dramabox-dit"},{"id":"dec","component":"codec-dramabox-vae"},{"id":"vocoder","component":"codec-bigvgan","detail":"The first BigVGAN turns stereo mel features into the low-rate waveform."},{"id":"bwe","component":"codec-dramabox-bwe"},{"id":"out","kind":"output","label":"Speech waveform"}],"edges":[{"from":"text","to":"gemma"},{"from":"gemma","to":"connector"},{"from":"connector","to":"flow","label":"Text cross-attention"},{"from":"ref","to":"mel","optional":true},{"from":"mel","to":"enc","optional":true},{"from":"enc","to":"flow","label":"Fixed reference latents","optional":true},{"from":"flow","to":"dec"},{"from":"dec","to":"vocoder"},{"from":"vocoder","to":"bwe"},{"from":"bwe","to":"out"}]}],"sources":["https://huggingface.co/ResembleAI/Dramabox","https://github.com/resemble-ai/DramaBox"],"packages":[{"id":"dramabox_q8_0","display_name":"DramaBox Q8_0 GGUF","precision":"q8_0"}],"docs":["docs/tts.md","docs/gguf.md"],"variants":["DramaBox"],"inputs":[{"type":"text","label":"Dialogue and delivery prompt","required":true},{"type":"audio","label":"Reference voice","required":false}],"detailStatus":"Prompt aggregation and connector, reference-latent conditioning, audio DiT, mel AudioVAE and residual bandwidth extension audited against the integration and official model description.","usageDoc":"docs/tts.md#dramabox"},{"id":"supertonic","family":"supertonic","name":"Supertonic 3","task":"speech-synthesis","tasks":["tts"],"summary":"Supertonic encodes characters with ConvNeXt and attention networks. Stored voice-style tensors condition sentence duration and latent flow generation. A causal ConvNeXt decoder directly predicts waveform patches; this path does not phonemize text or encode a reference recording.","routes":[{"name":"Preset-style TTS","nodes":[{"id":"text","kind":"input","label":"Text + language"},{"id":"style","kind":"input","label":"Stored voice style"},{"id":"tokens","component":"frontend-supertonic-characters"},{"id":"duration","component":"head-supertonic-duration"},{"id":"encoder","component":"encoder-supertonic-text"},{"id":"noise","kind":"input","label":"Gaussian noise"},{"id":"flow","component":"flow-supertonic-vector-field"},{"id":"decoder","component":"codec-supertonic-convnext"},{"id":"out","kind":"output","label":"Speech waveform"}],"edges":[{"from":"text","to":"tokens"},{"from":"tokens","to":"duration"},{"from":"tokens","to":"encoder"},{"from":"style","to":"duration"},{"from":"style","to":"encoder"},{"from":"duration","to":"flow","label":"Latent sequence length"},{"from":"encoder","to":"flow"},{"from":"style","to":"flow"},{"from":"noise","to":"flow"},{"from":"flow","to":"decoder"},{"from":"decoder","to":"out"}]}],"sources":["https://github.com/supertone-inc/supertonic"],"packages":[{"id":"supertonic_3_q8_0","display_name":"Supertonic 3 Q8_0 GGUF","precision":"q8_0"},{"id":"supertonic_3_f16","display_name":"Supertonic 3 F16 GGUF","precision":"f16"},{"id":"supertonic_3_orig","display_name":"Supertonic 3 Original-Dtype GGUF","precision":"orig"},{"id":"supertonic_3_safetensors","display_name":"Supertonic 3 Safetensors","precision":"native"}],"docs":["docs/tts.md","docs/gguf.md"],"variants":["Supertonic 3"],"inputs":[{"type":"text","label":"Text + language","required":true},{"type":"control","label":"Stored voice style / speed","required":false}],"detailStatus":"Character frontend, separate duration encoder, text/style cross-attention and waveform-patch decoder described.","usageDoc":"docs/tts.md#supertonic"},{"id":"kokoro_tts","family":"kokoro_tts","name":"Kokoro","task":"speech-synthesis","tasks":["tts"],"summary":"Kokoro uses PL-BERT for prosody and a separate convolutional/BiLSTM text encoder for synthesis. A packaged voice-style vector conditions duration, pitch and waveform generation. It does not run StyleTTS2's style diffusion or an audio-reference encoder.","routes":[{"name":"Preset-voice TTS","nodes":[{"id":"text","kind":"input","label":"Text / phonemes"},{"id":"voice","kind":"input","label":"Preset voice"},{"id":"g2p","component":"frontend-kokoro-phonemes","detail":"Supplied phonemes bypass grapheme-to-phoneme conversion."},{"id":"style","component":"encoder-preset-style"},{"id":"bert","component":"encoder-pl-bert"},{"id":"content","component":"encoder-styletts-text"},{"id":"predictor","component":"head-styletts-prosody"},{"id":"decoder","component":"codec-styletts-istft"},{"id":"out","kind":"output","label":"Speech waveform"}],"edges":[{"from":"text","to":"g2p"},{"from":"voice","to":"style"},{"from":"g2p","to":"style","label":"Phoneme count"},{"from":"g2p","to":"bert"},{"from":"g2p","to":"content"},{"from":"bert","to":"predictor"},{"from":"content","to":"predictor","label":"Text features"},{"from":"style","to":"predictor"},{"from":"predictor","to":"decoder","label":"Aligned features / pitch / noise"},{"from":"style","to":"decoder"},{"from":"decoder","to":"out"}]}],"sources":["https://huggingface.co/hexgrad/Kokoro-82M"],"packages":[{"id":"kokoro_82m_q8_0","display_name":"Kokoro 82M Q8_0 GGUF","precision":"q8_0"},{"id":"kokoro_82m_bf16","display_name":"Kokoro 82M BF16 GGUF","precision":"bf16"}],"docs":["tests/kokoro_tts/MULTILINGUAL_GGUF.md"],"variants":["Kokoro 82M"],"inputs":[{"type":"text","label":"Text or supplied phonemes","required":true},{"type":"control","label":"Preset voice / speed","required":false}],"detailStatus":"Parallel phoneme encoders, preset styles, duration expansion and source-conditioned ISTFT synthesis described.","usageDoc":"docs/models/kokoro_tts.md"},{"id":"kitten_tts","family":"kitten_tts","name":"KittenTTS","task":"speech-synthesis","tasks":["tts"],"summary":"KittenTTS uses eSpeak phonemes, a compact PL-BERT-style ALBERT encoder and a separate convolutional/BiLSTM text branch. Preset voice styles condition its duration/prosody predictor and StyleTTS2-derived inverse-STFT decoder.","routes":[{"name":"Preset-voice TTS","nodes":[{"id":"text","kind":"input","label":"English text"},{"id":"voice","kind":"input","label":"Preset voice"},{"id":"g2p","component":"frontend-espeak-phonemes"},{"id":"style","component":"encoder-preset-style"},{"id":"bert","component":"encoder-pl-bert"},{"id":"content","component":"encoder-styletts-text"},{"id":"predictor","component":"head-styletts-prosody"},{"id":"decoder","component":"codec-styletts-istft"},{"id":"out","kind":"output","label":"Speech waveform"}],"edges":[{"from":"text","to":"g2p"},{"from":"voice","to":"style"},{"from":"g2p","to":"style","label":"Phoneme count"},{"from":"g2p","to":"bert"},{"from":"g2p","to":"content"},{"from":"bert","to":"predictor"},{"from":"content","to":"predictor","label":"Text features"},{"from":"style","to":"predictor"},{"from":"predictor","to":"decoder","label":"Aligned features / pitch / noise"},{"from":"style","to":"decoder"},{"from":"decoder","to":"out"}]}],"sources":["https://huggingface.co/KittenML/kitten-tts-mini-0.8"],"packages":[{"id":"kitten_tts_mini_0_8_orig","display_name":"KittenTTS Mini 0.8 Original-Dtype GGUF","precision":"orig"}],"docs":["docs/community_models/kitten_tts.md","docs/tts.md","docs/gguf.md"],"variants":["KittenTTS Mini 0.8"],"inputs":[{"type":"text","label":"English text","required":true},{"type":"control","label":"Preset voice / speed","required":false}],"detailStatus":"eSpeak phonemes, shared ALBERT layers, duration/prosody and ISTFT decoder described.","usageDoc":"docs/community_models/kitten_tts.md"},{"id":"piper_tts","family":"piper_tts","name":"Piper TTS","task":"speech-synthesis","tasks":["tts"],"summary":"Piper's VITS inference encodes phonemes, samples durations and expands a text-conditioned latent prior. An invertible coupling network and a HiFi-GAN-style decoder synthesize the waveform. The packaged checkpoint supplies the voice; this path does not encode a cloning reference.","routes":[{"name":"VITS speech synthesis","nodes":[{"id":"text","kind":"input","label":"Text"},{"id":"g2p","component":"frontend-espeak-phonemes"},{"id":"encoder","component":"encoder-vits-text"},{"id":"duration","component":"head-vits-stochastic-duration"},{"id":"latent","component":"head-vits-alignment"},{"id":"flow","component":"flow-vits-coupling"},{"id":"decoder","component":"codec-vits-hifigan"},{"id":"out","kind":"output","label":"Speech waveform"}],"edges":[{"from":"text","to":"g2p"},{"from":"g2p","to":"encoder"},{"from":"encoder","to":"duration"},{"from":"encoder","to":"latent","label":"Prior mean / scale"},{"from":"duration","to":"latent","label":"Durations"},{"from":"latent","to":"flow"},{"from":"flow","to":"decoder"},{"from":"decoder","to":"out"}]}],"sources":["https://github.com/rhasspy/piper"],"packages":[{"id":"piper_lessac_medium_orig","display_name":"Piper Lessac Medium Original-Dtype GGUF","precision":"orig"}],"docs":["docs/tts.md","docs/community_models/piper_tts.md","docs/gguf.md"],"variants":["Piper Lessac Medium"],"inputs":[{"type":"text","label":"Text","required":true},{"type":"control","label":"Voice checkpoint / speed","required":false}],"detailStatus":"Phoneme encoder, stochastic duration flow and VITS inverse-coupling synthesis described.","usageDoc":"docs/community_models/piper_tts.md"},{"id":"inflect_v2","family":"inflect_v2","name":"Inflect Micro v2","task":"speech-synthesis","tasks":["tts"],"summary":"Inflect Micro v2 uses VITS-style latent synthesis with a deterministic convolutional duration predictor. A duration-expanded Gaussian prior is transformed by inverse coupling layers and decoded to audio; it is not autoregressive text generation.","routes":[{"name":"VITS speech synthesis","nodes":[{"id":"text","kind":"input","label":"Text"},{"id":"g2p","component":"frontend-espeak-phonemes"},{"id":"encoder","component":"encoder-vits-text"},{"id":"duration","component":"head-vits-duration"},{"id":"latent","component":"head-vits-alignment"},{"id":"flow","component":"flow-vits-coupling"},{"id":"decoder","component":"codec-vits-hifigan"},{"id":"out","kind":"output","label":"Speech waveform"}],"edges":[{"from":"text","to":"g2p"},{"from":"g2p","to":"encoder"},{"from":"encoder","to":"duration"},{"from":"encoder","to":"latent","label":"Prior mean / scale"},{"from":"duration","to":"latent","label":"Durations"},{"from":"latent","to":"flow"},{"from":"flow","to":"decoder"},{"from":"decoder","to":"out"}]}],"sources":["https://huggingface.co/owensong/Inflect-Micro-v2"],"packages":[{"id":"inflect_micro_v2_orig","display_name":"Inflect Micro v2 Original-Dtype GGUF","precision":"orig"}],"docs":["docs/tts.md","docs/community_models/inflect_v2.md","docs/gguf.md"],"variants":["Inflect Micro v2"],"inputs":[{"type":"text","label":"Text","required":true},{"type":"control","label":"Speed / variation","required":false}],"detailStatus":"VITS phoneme encoder with deterministic convolutional duration prediction described.","usageDoc":"docs/community_models/inflect_v2.md"},{"id":"sanotts","family":"sanotts","name":"SanoTTS","task":"speech-synthesis","tasks":["tts"],"summary":"SanoTTS distills speech generation into duration and acoustic convolutional students. Nano predicts mel features for spectral synthesis; PiperLite predicts continuous decoder latents. Their teacher lineage does not mean they retain Kokoro's PL-BERT or Piper's full VITS inference network.","routes":[{"name":"Nano","nodes":[{"id":"text","kind":"input","label":"Text"},{"id":"g2p","component":"frontend-espeak-phonemes","detail":"Nano-specific phoneme remapping and vocabulary."},{"id":"student","component":"head-sano-acoustic-student","detail":"Frame output is a 100-bin mel representation."},{"id":"decoder","component":"codec-sano-convnext-istft"},{"id":"out","kind":"output","label":"Speech waveform"}],"edges":[{"from":"text","to":"g2p"},{"from":"g2p","to":"student"},{"from":"student","to":"decoder"},{"from":"decoder","to":"out"}]},{"name":"PiperLite","nodes":[{"id":"text","kind":"input","label":"Text"},{"id":"g2p","component":"frontend-espeak-phonemes","detail":"Piper-style phoneme ID mapping, distinct from Nano's mapping."},{"id":"student","component":"head-sano-acoustic-student","detail":"Frame output is a continuous acoustic latent, not a mel spectrogram."},{"id":"decoder","component":"codec-sano-residual-decoder"},{"id":"out","kind":"output","label":"Speech waveform"}],"edges":[{"from":"text","to":"g2p"},{"from":"g2p","to":"student"},{"from":"student","to":"decoder"},{"from":"decoder","to":"out"}]}],"sources":["https://huggingface.co/ampixa/sanoTTS"],"packages":[{"id":"sanotts_heart_nano_orig","display_name":"sanoTTS heart-nano 294k FP32 GGUF","precision":"orig"},{"id":"sanotts_heart_orig","display_name":"sanoTTS heart 2.27M FP32 GGUF","precision":"orig"},{"id":"sanotts_amy_orig","display_name":"sanoTTS amy 1.46M FP32 GGUF (English, piperlite)","precision":"orig"},{"id":"sanotts_hfc_orig","display_name":"sanoTTS hfc 1.83M FP32 GGUF (English, piperlite)","precision":"orig"},{"id":"sanotts_kristin_orig","display_name":"sanoTTS kristin 1.40M FP32 GGUF (English, piperlite)","precision":"orig"},{"id":"sanotts_vi_orig","display_name":"sanoTTS vi 1.57M FP32 GGUF (Vietnamese, piperlite)","precision":"orig"},{"id":"sanotts_id_orig","display_name":"sanoTTS id 1.56M FP32 GGUF (Indonesian, piperlite)","precision":"orig"},{"id":"sanotts_cs_orig","display_name":"sanoTTS cs 1.57M FP32 GGUF (Czech, piperlite)","precision":"orig"},{"id":"sanotts_de_orig","display_name":"sanoTTS de 1.57M FP32 GGUF (German, piperlite)","precision":"orig"},{"id":"sanotts_es_orig","display_name":"sanoTTS es 1.56M FP32 GGUF (Spanish, piperlite)","precision":"orig"},{"id":"sanotts_fr_orig","display_name":"sanoTTS fr 1.57M FP32 GGUF (French, piperlite)","precision":"orig"},{"id":"sanotts_it_orig","display_name":"sanoTTS it 1.57M FP32 GGUF (Italian, piperlite)","precision":"orig"},{"id":"sanotts_pt_orig","display_name":"sanoTTS pt 1.57M FP32 GGUF (Portuguese (Brazil), piperlite)","precision":"orig"},{"id":"sanotts_ro_orig","display_name":"sanoTTS ro 1.57M FP32 GGUF (Romanian, piperlite)","precision":"orig"},{"id":"sanotts_ru_orig","display_name":"sanoTTS ru 1.57M FP32 GGUF (Russian, piperlite)","precision":"orig"},{"id":"sanotts_tr_orig","display_name":"sanoTTS tr 1.56M FP32 GGUF (Turkish, piperlite)","precision":"orig"},{"id":"sanotts_ne_orig","display_name":"sanoTTS ne 1.47M FP32 GGUF (Nepali, piperlite)","precision":"orig"},{"id":"sanotts_hi_orig","display_name":"sanoTTS hi 1.50M FP32 GGUF (Hindi, piperlite)","precision":"orig"}],"docs":["docs/tts.md","docs/community_models/sanotts.md","docs/gguf.md"],"variants":["sanoTTS heart-nano 294k","sanoTTS heart 2.27M","sanoTTS amy 1.46M (English, piperlite)","sanoTTS hfc 1.83M (English, piperlite)","sanoTTS kristin 1.40M (English, piperlite)","sanoTTS vi 1.57M (Vietnamese, piperlite)","sanoTTS id 1.56M (Indonesian, piperlite)","sanoTTS cs 1.57M (Czech, piperlite)","sanoTTS de 1.57M (German, piperlite)","sanoTTS es 1.56M (Spanish, piperlite)","sanoTTS fr 1.57M (French, piperlite)","sanoTTS it 1.57M (Italian, piperlite)","sanoTTS pt 1.57M (Portuguese (Brazil), piperlite)","sanoTTS ro 1.57M (Romanian, piperlite)","sanoTTS ru 1.57M (Russian, piperlite)","sanoTTS tr 1.56M (Turkish, piperlite)","sanoTTS ne 1.47M (Nepali, piperlite)","sanoTTS hi 1.50M (Hindi, piperlite)"],"inputs":[{"type":"text","label":"Text","required":true},{"type":"control","label":"Voice checkpoint / speed","required":false}],"detailStatus":"Separate Nano and PiperLite student pipelines described; teacher architectures are not substituted for the student.","usageDoc":"docs/community_models/sanotts.md"},{"id":"index_tts2","family":"index_tts2","name":"IndexTTS2","task":"speech-synthesis","tasks":["tts","clone"],"summary":"GPT-style AR predicts semantic speech codes with separate speaker and emotion conditioning. S2Mel flow and BigVGAN synthesize speech. Version 2 uses semantic speaker prompts and AR latents; 2.5 uses CAM++ speaker prompts and an enhanced code decoder. Optional Qwen3 converts emotion text into control weights, not speech tokens.","routes":[{"name":"IndexTTS 2","nodes":[{"id":"text","kind":"input","label":"Speech text"},{"id":"ref","kind":"input","label":"Reference voice"},{"id":"emotion","kind":"input","label":"Optional emotion audio / text / vector","optional":true},{"id":"tok","component":"frontend-sentencepiece"},{"id":"semantic","component":"encoder-w2v-bert"},{"id":"condition","component":"encoder-index-condition","detail":"Separate speaker and emotion encoders. Emotion audio defaults to the voice reference; text can be converted to weights by optional Qwen3."},{"id":"emotiontext","component":"encoder-index-emotion"},{"id":"speaker","component":"encoder-campplus"},{"id":"mel","component":"frontend-logmel"},{"id":"ar","component":"ar-gpt"},{"id":"codes","component":"encoder-index-semantic-vq","detail":"Reference features are quantized; generated codes are embedded. Projected AR hidden states are added only to generated content."},{"id":"flow","component":"flow-s2mel"},{"id":"vocoder","component":"codec-bigvgan"},{"id":"out","kind":"output","label":"Speech waveform"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"ar"},{"from":"ref","to":"semantic"},{"from":"semantic","to":"condition"},{"from":"emotion","to":"condition","optional":true,"label":"Audio through Wav2Vec2-BERT / explicit weights"},{"from":"emotion","to":"emotiontext","optional":true,"label":"Text control only"},{"from":"emotiontext","to":"condition","optional":true,"label":"Emotion weights"},{"from":"condition","to":"ar"},{"from":"ref","to":"speaker"},{"from":"ref","to":"mel"},{"from":"semantic","to":"codes"},{"from":"ar","to":"codes","label":"Generated codes + hidden states"},{"from":"codes","to":"flow"},{"from":"speaker","to":"flow"},{"from":"mel","to":"flow"},{"from":"flow","to":"vocoder"},{"from":"vocoder","to":"out"}]},{"name":"IndexTTS 2.5","nodes":[{"id":"text","kind":"input","label":"Speech text + language"},{"id":"ref","kind":"input","label":"Reference voice"},{"id":"emotion","kind":"input","label":"Optional emotion audio / text / vector","optional":true},{"id":"tok","component":"frontend-sentencepiece"},{"id":"semantic","component":"encoder-w2v-bert"},{"id":"condition","component":"encoder-index-condition","detail":"Emotion conditioning only; speaker identity comes from CAM++. Emotion audio defaults to the voice reference."},{"id":"emotiontext","component":"encoder-index-emotion"},{"id":"speaker","component":"encoder-campplus"},{"id":"mel","component":"frontend-logmel"},{"id":"ar","component":"ar-gpt"},{"id":"codes","component":"encoder-index-enhanced-codec"},{"id":"flow","component":"flow-s2mel"},{"id":"vocoder","component":"codec-bigvgan"},{"id":"out","kind":"output","label":"Speech waveform"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"ar"},{"from":"ref","to":"semantic"},{"from":"semantic","to":"condition"},{"from":"emotion","to":"condition","optional":true,"label":"Audio through Wav2Vec2-BERT / explicit weights"},{"from":"emotion","to":"emotiontext","optional":true,"label":"Text control only"},{"from":"emotiontext","to":"condition","optional":true,"label":"Emotion weights"},{"from":"condition","to":"ar"},{"from":"ref","to":"speaker"},{"from":"speaker","to":"ar","label":"Speaker token"},{"from":"ref","to":"mel"},{"from":"semantic","to":"flow","label":"Unquantized reference features"},{"from":"ar","to":"codes","label":"Generated codes only"},{"from":"codes","to":"flow"},{"from":"speaker","to":"flow"},{"from":"mel","to":"flow"},{"from":"flow","to":"vocoder"},{"from":"vocoder","to":"out"}]}],"sources":["https://github.com/index-tts/index-tts","https://huggingface.co/IndexTeam/IndexTTS-2.5"],"packages":[{"id":"index_tts2_q8_0","display_name":"IndexTTS2 Q8_0 GGUF","precision":"q8_0"},{"id":"index_tts2_f16","display_name":"IndexTTS2 F16 GGUF","precision":"f16"},{"id":"index_tts2_orig","display_name":"IndexTTS2 Original-Dtype GGUF","precision":"orig"},{"id":"index_tts2_safetensors","display_name":"IndexTTS2 Safetensors","precision":"native"},{"id":"index_tts2_5_q8_0","display_name":"IndexTTS2.5 Q8_0 GGUF","precision":"q8_0"},{"id":"index_tts2_5_f16","display_name":"IndexTTS2.5 F16 GGUF","precision":"f16"},{"id":"index_tts2_5_orig","display_name":"IndexTTS2.5 Original-Dtype GGUF","precision":"orig"}],"docs":["docs/tts.md","docs/gguf.md"],"variants":["IndexTTS2","IndexTTS2.5"],"inputs":[{"type":"text","label":"Speech text","required":true},{"type":"audio","label":"Reference voice","required":true},{"type":"audio","label":"Emotion reference; defaults to voice reference","required":false},{"type":"text","label":"Emotion description or explicit emotion vector","required":false}],"detailStatus":"Version-specific speaker, emotion, semantic-code and acoustic branches audited.","usageDoc":"docs/models/index_tts.md"},{"id":"confucius4_tts","family":"confucius4_tts","name":"Confucius4-TTS","task":"speech-synthesis","tasks":["clone"],"summary":"A GPT-style AR Transformer predicts semantic codes and hidden states from text and a Wav2Vec2-BERT-derived speaker prompt. A conditional flow model combines these with reference mel and CAM++ identity, then BigVGAN synthesizes speech. Reference transcription is not required.","routes":[{"name":"Reference-conditioned speech","nodes":[{"id":"text","kind":"input","label":"Speech text + language"},{"id":"ref","kind":"input","label":"Reference voice"},{"id":"tok","component":"frontend-sentencepiece"},{"id":"semantic","component":"encoder-w2v-bert"},{"id":"prompt","component":"encoder-confucius-ecapa"},{"id":"speaker","component":"encoder-campplus"},{"id":"mel","component":"frontend-logmel"},{"id":"ar","component":"ar-t2s-transformer"},{"id":"flow","component":"flow-s2a"},{"id":"vocoder","component":"codec-bigvgan"},{"id":"out","kind":"output","label":"Speech waveform"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"ar"},{"from":"ref","to":"semantic"},{"from":"semantic","to":"prompt"},{"from":"prompt","to":"ar"},{"from":"ref","to":"speaker"},{"from":"ref","to":"mel"},{"from":"speaker","to":"flow"},{"from":"mel","to":"flow"},{"from":"ar","to":"flow","label":"Semantic codes + hidden states"},{"from":"flow","to":"vocoder"},{"from":"vocoder","to":"out"}]}],"sources":["https://huggingface.co/netease-youdao/Confucius4-TTS"],"packages":[{"id":"confucius4_tts_orig","display_name":"Confucius4-TTS Original-Dtype GGUF","precision":"orig"}],"docs":["docs/tts.md","docs/gguf.md"],"variants":["Confucius4-TTS"],"inputs":[{"type":"text","label":"Speech text + language","required":true},{"type":"audio","label":"Reference voice","required":true}],"detailStatus":"Reference semantic, speaker-style and mel branches audited.","usageDoc":"docs/tts.md#confucius4-tts"},{"id":"magpie_tts","family":"magpie_tts","name":"MagpieTTS Multilingual 357M","task":"speech-synthesis","tasks":["tts"],"summary":"MagpieTTS uses a causal text encoder, a text-cross-attending temporal AR decoder and a separate local AR Transformer for frame-stacked codec prediction. NanoCodec converts finite scalar codes to waveform audio. Voices come from stored context embeddings, not a reference-audio encoder.","routes":[{"name":"Preset-voice multilingual TTS","nodes":[{"id":"text","kind":"input","label":"Text + language"},{"id":"voice","kind":"input","label":"Preset speaker ID"},{"id":"front","component":"frontend-magpie-text"},{"id":"encoder","component":"encoder-magpie-text"},{"id":"preset","component":"encoder-magpie-preset"},{"id":"ar","component":"ar-magpie-temporal"},{"id":"local","component":"ar-magpie-local"},{"id":"codec","component":"codec-nemo-audio-codec"},{"id":"out","kind":"output","label":"Speech waveform"}],"edges":[{"from":"text","to":"front"},{"from":"front","to":"encoder"},{"from":"voice","to":"preset"},{"from":"encoder","to":"ar","label":"Text memory"},{"from":"preset","to":"ar"},{"from":"ar","to":"local"},{"from":"local","to":"codec","label":"Codec IDs"},{"from":"codec","to":"out"}]}],"sources":["https://huggingface.co/nvidia/magpie_tts_multilingual_357m"],"packages":[{"id":"magpie_tts_orig","display_name":"MagpieTTS Multilingual 357M Original-Dtype GGUF","precision":"orig"},{"id":"magpie_tts_q8_0","display_name":"MagpieTTS Multilingual 357M Q8_0 GGUF","precision":"q8_0"}],"docs":["docs/models/magpie_tts.md","docs/tts.md","docs/gguf.md"],"variants":["MagpieTTS Multilingual 357M"],"inputs":[{"type":"text","label":"Target text","required":true},{"type":"control","label":"Language / preset voice","required":false}],"detailStatus":"Language-specific tokenization, causal text encoding, temporal/local AR distinction, preset voice context and NanoCodec audited. Reference cloning is not exposed by this integration.","usageDoc":"docs/models/magpie_tts.md"},{"id":"higgs_audio_stt","family":"higgs_audio_stt","name":"Higgs Audio STT","task":"speech-recognition-audio-understanding","tasks":["asr"],"summary":"Whisper-Large-v3 audio features pass through temporal downsampling and an MLP adapter into Qwen3. The Whisper text decoder is not used.","routes":[{"name":"Speech recognition","nodes":[{"id":"audio","kind":"input","label":"Speech audio"},{"id":"text","kind":"input","label":"Optional text context","optional":true},{"id":"mel","component":"frontend-logmel"},{"id":"encoder","component":"encoder-whisper","detail":"Whisper-Large-v3 encoder architecture."},{"id":"adapter","component":"encoder-higgs-stt-projector"},{"id":"tokenizer","component":"frontend-bpe"},{"id":"ar","component":"ar-qwen3","detail":"Qwen3-1.7B text decoder."},{"id":"out","kind":"output","label":"Transcript"}],"edges":[{"from":"audio","to":"mel"},{"from":"mel","to":"encoder"},{"from":"encoder","to":"adapter"},{"from":"adapter","to":"ar"},{"from":"text","to":"tokenizer","optional":true},{"from":"tokenizer","to":"ar","optional":true},{"from":"ar","to":"out"}]}],"sources":["https://huggingface.co/bosonai/higgs-audio-v3-stt"],"packages":[{"id":"higgs_audio_stt_q8_0","display_name":"Higgs Audio v3 STT Q8_0 GGUF","precision":"q8_0"},{"id":"higgs_audio_stt_f16","display_name":"Higgs Audio v3 STT F16 GGUF","precision":"f16"}],"docs":["docs/models/higgs_audio_stt.md","docs/asr.md","docs/gguf.md"],"variants":["Higgs Audio v3 STT"],"inputs":[{"type":"audio","label":"Speech","required":true},{"type":"text","label":"Context / language hint","required":false}],"detailStatus":"Audio adapter and optional text-context branches audited.","usageDoc":"docs/models/higgs_audio_stt.md"},{"id":"moss_transcribe_diarize","family":"moss_transcribe_diarize","name":"MOSS-Transcribe-Diarize","task":"speech-recognition-audio-understanding","tasks":["asr"],"summary":"Whisper features are stacked and projected into Qwen3. Instruction tokens and time markers guide generation of a serialized transcript with speaker and timing information; there is no separate speaker-clustering model in this core route.","routes":[{"name":"Transcription / diarization text","nodes":[{"id":"audio","kind":"input","label":"Speech audio"},{"id":"text","kind":"input","label":"Instruction","optional":true},{"id":"mel","component":"frontend-logmel"},{"id":"encoder","component":"encoder-whisper"},{"id":"adapter","component":"encoder-moss-stt-projector"},{"id":"tokenizer","component":"frontend-bpe"},{"id":"ar","component":"ar-qwen3"},{"id":"out","kind":"output","label":"Text + speaker / time labels"}],"edges":[{"from":"audio","to":"mel"},{"from":"mel","to":"encoder"},{"from":"encoder","to":"adapter"},{"from":"adapter","to":"ar","label":"Audio embeddings"},{"from":"text","to":"tokenizer","optional":true},{"from":"tokenizer","to":"ar","label":"Prompt / time markers"},{"from":"ar","to":"out"}]}],"sources":["https://huggingface.co/OpenMOSS-Team/MOSS-Transcribe-Diarize/blob/main/config.json"],"packages":[{"id":"moss_transcribe_diarize_bf16","display_name":"MOSS-Transcribe-Diarize GGUF BF16","precision":"bf16"},{"id":"moss_transcribe_diarize_q8_0","display_name":"MOSS-Transcribe-Diarize GGUF Q8_0","precision":"q8_0"},{"id":"moss_transcribe_diarize_q4_k","display_name":"MOSS-Transcribe-Diarize GGUF Q4_K","precision":"q4_k"}],"docs":[],"variants":["MOSS-Transcribe-Diarize"],"inputs":[{"type":"audio","label":"Speech","required":true},{"type":"text","label":"Instruction","required":false}],"detailStatus":"Acoustic projection and instruction branches audited.","usageDoc":"docs/models/moss_transcribe_diarize.md"},{"id":"fun_asr_nano","family":"fun_asr_nano","name":"Fun-ASR-Nano","task":"speech-recognition-audio-understanding","tasks":["asr"],"summary":"A SenseVoice-style SANM encoder processes filterbank features. A two-layer Transformer adapter maps them into Qwen3-0.6B for autoregressive transcription.","routes":[{"name":"Speech recognition","nodes":[{"id":"audio","kind":"input","label":"Speech audio"},{"id":"text","kind":"input","label":"Prompt / language","optional":true},{"id":"features","component":"frontend-fbank-lfr"},{"id":"encoder","component":"encoder-sensevoice-sanm"},{"id":"adapter","component":"encoder-fun-transformer-adapter"},{"id":"tokenizer","component":"frontend-bpe"},{"id":"ar","component":"ar-qwen3","detail":"Qwen3-0.6B decoder."},{"id":"out","kind":"output","label":"Transcript"}],"edges":[{"from":"audio","to":"features"},{"from":"features","to":"encoder"},{"from":"encoder","to":"adapter"},{"from":"adapter","to":"ar"},{"from":"text","to":"tokenizer","optional":true},{"from":"tokenizer","to":"ar","optional":true},{"from":"ar","to":"out"}]}],"sources":["https://huggingface.co/FunAudioLLM/Fun-ASR-Nano-2512/blob/main/config.yaml"],"packages":[{"id":"fun_asr_nano_2512_q8_0","display_name":"Fun-ASR-Nano-2512 Q8_0 GGUF","precision":"q8_0"},{"id":"fun_asr_nano_2512_f16","display_name":"Fun-ASR-Nano-2512 F16 GGUF","precision":"f16"},{"id":"fun_asr_nano_2512_safetensors","display_name":"Fun-ASR-Nano-2512 HF Safetensors","precision":"native"}],"docs":["docs/asr.md","docs/gguf.md"],"variants":["Fun-ASR-Nano-2512"],"inputs":[{"type":"audio","label":"Speech","required":true},{"type":"text","label":"Prompt / language","required":false}],"detailStatus":"SANM, Transformer adapter and text-prompt branches audited.","usageDoc":"docs/models/fun_asr_nano.md"},{"id":"qwen3_asr","family":"qwen3_asr","name":"Qwen3-ASR","task":"speech-recognition-audio-understanding","tasks":["asr"],"summary":"Log-mel features pass through a dedicated audio Transformer. Projected audio embeddings and text context condition a Qwen3 transcript decoder.","routes":[{"name":"Speech recognition","nodes":[{"id":"audio","kind":"input","label":"Audio"},{"id":"text","kind":"input","label":"Optional text context","optional":true},{"id":"mel","component":"frontend-qwen-mel"},{"id":"enc","component":"encoder-qwen3-asr-encoder"},{"id":"tok","component":"frontend-bpe"},{"id":"ar","component":"ar-qwen3","detail":"Qwen3-ASR text decoder; model-specific rotary positions."},{"id":"out","kind":"output","label":"Transcript + language"}],"edges":[{"from":"audio","to":"mel"},{"from":"mel","to":"enc"},{"from":"text","to":"tok","optional":true},{"from":"tok","to":"ar","optional":true},{"from":"enc","to":"ar","label":"Audio embeddings"},{"from":"ar","to":"out"}]}],"sources":["https://huggingface.co/Qwen/Qwen3-ASR-0.6B"],"packages":[{"id":"qwen3_asr_1_7b_q8_0","display_name":"Qwen3-ASR 1.7B Q8_0 GGUF","precision":"q8_0"},{"id":"qwen3_asr_1_7b_f16","display_name":"Qwen3-ASR 1.7B F16 GGUF","precision":"f16"},{"id":"qwen3_asr_0_6b_q8_0","display_name":"Qwen3-ASR 0.6B Q8_0 GGUF","precision":"q8_0"},{"id":"qwen3_asr_0_6b_f16","display_name":"Qwen3-ASR 0.6B F16 GGUF","precision":"f16"},{"id":"qwen3_asr_1_7b_safetensors","display_name":"Qwen3-ASR 1.7B HF Safetensors","precision":"native"},{"id":"qwen3_asr_0_6b_safetensors","display_name":"Qwen3-ASR 0.6B Safetensors","precision":"native"}],"docs":["docs/models/qwen3.md","docs/asr.md","docs/gguf.md"],"variants":["0.6B","1.7B"],"inputs":[{"type":"audio","label":"Audio","required":true},{"type":"text","label":"Context / language hint","required":false}],"detailStatus":"Audio and text conditioning branches audited.","usageDoc":"docs/models/qwen3.md#qwen3-asr"},{"id":"confucius4_r2t2","family":"confucius4_r2t2","name":"Confucius4-R2T2","task":"speech-recognition-audio-understanding","tasks":["asr"],"summary":"Confucius4-R2T2 adapts the Qwen3-ASR architecture for streaming recognition: windowed acoustic attention and a projected audio sequence condition a Qwen3 text decoder.","routes":[{"name":"Speech recognition","nodes":[{"id":"audio","kind":"input","label":"Speech audio"},{"id":"text","kind":"input","label":"Context / language","optional":true},{"id":"mel","component":"frontend-qwen-mel"},{"id":"encoder","component":"encoder-qwen3-asr-encoder","detail":"R2T2's acoustic attention window and streaming policy; separate model weights."},{"id":"tok","component":"frontend-bpe"},{"id":"decoder","component":"ar-qwen3"},{"id":"out","kind":"output","label":"Transcript"}],"edges":[{"from":"audio","to":"mel"},{"from":"mel","to":"encoder"},{"from":"encoder","to":"decoder"},{"from":"text","to":"tok","optional":true},{"from":"tok","to":"decoder","optional":true},{"from":"decoder","to":"out"}]}],"sources":["https://huggingface.co/netease-youdao/Confucius4-R2T2"],"packages":[{"id":"confucius4_r2t2_q8_0","display_name":"Confucius4-R2T2 Q8_0 GGUF","precision":"q8_0"},{"id":"confucius4_r2t2_f16","display_name":"Confucius4-R2T2 F16 GGUF","precision":"f16"},{"id":"confucius4_r2t2_q4_k_m","display_name":"Confucius4-R2T2 Q4_K_M GGUF","precision":"q4_k_m"},{"id":"confucius4_r2t2_safetensors","display_name":"Confucius4-R2T2 (HF safetensors)","precision":"native"}],"docs":["docs/community_models/r2t2.md","docs/asr.md"],"variants":["Confucius4-R2T2"],"inputs":[{"type":"audio","label":"Speech","required":true},{"type":"text","label":"Context / language hint","required":false}],"detailStatus":"Qwen3-ASR-derived audio tower and text-conditioning branches audited.","usageDoc":"docs/community_models/r2t2.md"},{"id":"audio8_asr","family":"audio8_asr","name":"Audio8 ASR","task":"speech-recognition-audio-understanding","tasks":["asr"],"summary":"A Qwen3-ASR audio encoder supplies features to a residual MLP tower and adaptive pooling adapter. A small Qwen2-family decoder generates text; the transcription instruction is fixed in this route.","routes":[{"name":"Speech recognition","nodes":[{"id":"audio","kind":"input","label":"Speech audio"},{"id":"mel","component":"frontend-qwen-mel"},{"id":"encoder","component":"encoder-qwen3-asr-encoder"},{"id":"adapter","component":"encoder-audio8-mlp-adapter"},{"id":"decoder","component":"ar-qwen2","detail":"Fixed transcription instruction plus projected audio embeddings."},{"id":"out","kind":"output","label":"Transcript"}],"edges":[{"from":"audio","to":"mel"},{"from":"mel","to":"encoder"},{"from":"encoder","to":"adapter"},{"from":"adapter","to":"decoder"},{"from":"decoder","to":"out"}]}],"sources":["https://huggingface.co/Edge0/Audio8-ASR-0.1B"],"packages":[{"id":"audio8_asr_0_1b_q8_0","display_name":"Audio8-ASR-0.1B Q8_0 GGUF","precision":"q8_0"},{"id":"audio8_asr_0_1b_f16","display_name":"Audio8-ASR-0.1B F16 GGUF","precision":"f16"},{"id":"audio8_asr_0_1b_safetensors","display_name":"Audio8-ASR-0.1B HF Safetensors","precision":"bfloat16"}],"docs":["docs/asr.md","docs/community_models/audio8_asr.md"],"variants":["Audio8-ASR-0.1B"],"inputs":[{"type":"audio","label":"Speech","required":true}],"detailStatus":"Audio tower, residual MLP adapter and fixed transcription prompt audited.","usageDoc":"docs/community_models/audio8_asr.md"},{"id":"vibevoice_asr","family":"vibevoice_asr","name":"VibeVoice ASR","task":"speech-recognition-audio-understanding","tasks":["asr"],"summary":"Acoustic and semantic waveform encoders feed separate connectors whose outputs are summed. Qwen2-family autoregressive decoding generates structured transcription, including speaker and timing information.","routes":[{"name":"Structured transcription","nodes":[{"id":"audio","kind":"input","label":"Speech audio"},{"id":"text","kind":"input","label":"Context / hotwords","optional":true},{"id":"encoder","component":"encoder-vibevoice-tokenizers"},{"id":"tok","component":"frontend-bpe"},{"id":"decoder","component":"ar-qwen2"},{"id":"out","kind":"output","label":"Text + speaker / time labels"}],"edges":[{"from":"audio","to":"encoder"},{"from":"encoder","to":"decoder","label":"Summed speech embeddings"},{"from":"text","to":"tok","optional":true},{"from":"tok","to":"decoder","optional":true},{"from":"decoder","to":"out"}]}],"sources":["https://huggingface.co/microsoft/VibeVoice-ASR"],"packages":[{"id":"vibevoice_asr_q8_0","display_name":"VibeVoice ASR Q8_0 GGUF","precision":"q8_0"},{"id":"vibevoice_asr_f16","display_name":"VibeVoice ASR F16 GGUF","precision":"f16"}],"docs":["docs/asr.md","docs/gguf.md"],"variants":["VibeVoice ASR"],"inputs":[{"type":"audio","label":"Speech","required":true},{"type":"text","label":"Context / hotwords","required":false}],"detailStatus":"Dual speech encoders, feature fusion and text-context branches audited.","usageDoc":"docs/models/vibevoice_asr.md#vibevoice-asr"},{"id":"vibevoice_asr_streaming","family":"vibevoice_asr_streaming","name":"VibeVoice ASR Streaming","task":"speech-recognition-audio-understanding","tasks":["asr"],"summary":"The streaming VibeVoice checkpoint shares the dual waveform-encoder design and Qwen2 architecture. Audio chunks and generated text are interleaved in the decoder context; it is not the offline checkpoint run with a different display mode.","routes":[{"name":"Streaming speaker-attributed transcription","nodes":[{"id":"audio","kind":"input","label":"Speech chunks"},{"id":"text","kind":"input","label":"Context / hotwords","optional":true},{"id":"encoder","component":"encoder-vibevoice-tokenizers"},{"id":"tok","component":"frontend-bpe"},{"id":"decoder","component":"ar-qwen2","detail":"Streaming checkpoint; audio and generated-text chunks share the evolving decoder context."},{"id":"out","kind":"output","label":"Incremental speaker-attributed text"}],"edges":[{"from":"audio","to":"encoder"},{"from":"encoder","to":"decoder","label":"Speech embeddings"},{"from":"text","to":"tok","optional":true},{"from":"tok","to":"decoder","optional":true},{"from":"decoder","to":"out"}]}],"sources":["https://huggingface.co/microsoft/VibeVoice-ASR-Streaming-7B"],"packages":[{"id":"vibevoice_asr_streaming_7b_q8_0","display_name":"VibeVoice ASR Streaming 7B Q8_0 GGUF","precision":"q8_0"},{"id":"vibevoice_asr_streaming_7b_bf16","display_name":"VibeVoice ASR Streaming 7B BF16 GGUF","precision":"bf16"},{"id":"vibevoice_asr_streaming_7b_q4_k","display_name":"VibeVoice ASR Streaming 7B Q4_K GGUF","precision":"q4_k"},{"id":"vibevoice_asr_streaming_1_5b_q8_0","display_name":"VibeVoice ASR Streaming 1.5B Q8_0 GGUF","precision":"q8_0"},{"id":"vibevoice_asr_streaming_1_5b_bf16","display_name":"VibeVoice ASR Streaming 1.5B BF16 GGUF","precision":"bf16"},{"id":"vibevoice_asr_streaming_1_5b_q4_k","display_name":"VibeVoice ASR Streaming 1.5B Q4_K GGUF","precision":"q4_k"}],"docs":["docs/asr.md","docs/gguf.md"],"variants":["VibeVoice ASR Streaming 7B","VibeVoice ASR Streaming 1.5B"],"inputs":[{"type":"audio","label":"Speech stream","required":true},{"type":"text","label":"Context / hotwords","required":false}],"detailStatus":"Dual encoder and interleaved audio/text route audited.","usageDoc":"docs/models/vibevoice_asr.md#vibevoice-asr-streaming-7b"},{"id":"vibeasr","family":"vibeasr","name":"VibeASR (BitNet)","task":"speech-recognition-audio-understanding","tasks":["asr"],"summary":"VibeASR combines two convolutional waveform encoders with a Qwen2-family decoder using BitNet-style quantized projections. The published encoder uses ReLU feed-forward blocks rather than the original VibeVoice tokenizer's GELU blocks.","routes":[{"name":"Compact structured transcription","nodes":[{"id":"audio","kind":"input","label":"Speech audio"},{"id":"text","kind":"input","label":"Context","optional":true},{"id":"encoder","component":"encoder-vibeasr-vae-encoder"},{"id":"tok","component":"frontend-bpe"},{"id":"decoder","component":"ar-bitnet-language-model"},{"id":"out","kind":"output","label":"Transcript / structured text"}],"edges":[{"from":"audio","to":"encoder"},{"from":"encoder","to":"decoder","label":"Summed speech embeddings"},{"from":"text","to":"tok","optional":true},{"from":"tok","to":"decoder","optional":true},{"from":"decoder","to":"out"}]}],"sources":["https://huggingface.co/microsoft/VibeVoice-ASR-BitNet"],"packages":[{"id":"vibeasr_bitnet_i2_s","display_name":"VibeVoice-ASR-BitNet I8_S encoder + I2_S decoder","precision":"native"}],"docs":["docs/community_models/vibeasr.md"],"variants":["VibeVoice-ASR-BitNet"],"inputs":[{"type":"audio","label":"Speech","required":true},{"type":"text","label":"Context","required":false}],"detailStatus":"Published quantized encoder and Qwen2-family decoder route audited.","usageDoc":"docs/community_models/vibeasr.md"},{"id":"voxtral_realtime","family":"voxtral_realtime","name":"Voxtral Mini 4B Realtime","task":"speech-recognition-audio-understanding","tasks":["asr"],"summary":"Voxtral Realtime encodes log-mel audio with a causal Transformer, projects grouped frames and adds them to text-token embeddings. Delay-conditioned normalization modulates its autoregressive decoder.","routes":[{"name":"Realtime transcription","nodes":[{"id":"audio","kind":"input","label":"Speech stream"},{"id":"delay","kind":"input","label":"Transcription delay","optional":true},{"id":"mel","component":"frontend-logmel"},{"id":"encoder","component":"encoder-voxtral-audio-encoder"},{"id":"decoder","component":"ar-voxtral-decoder"},{"id":"out","kind":"output","label":"Incremental transcript"}],"edges":[{"from":"audio","to":"mel"},{"from":"mel","to":"encoder"},{"from":"encoder","to":"decoder","label":"Projected audio embeddings"},{"from":"delay","to":"decoder","label":"Delay embedding","optional":true},{"from":"decoder","to":"out"}]}],"sources":["https://huggingface.co/mistralai/Voxtral-Mini-4B-Realtime-2602"],"packages":[{"id":"voxtral_realtime_q8_0","display_name":"Voxtral Mini 4B Realtime Q8_0 GGUF","precision":"q8_0"},{"id":"voxtral_realtime_q4_k","display_name":"Voxtral Mini 4B Realtime Q4_K GGUF","precision":"q4_k"},{"id":"voxtral_realtime_bf16","display_name":"Voxtral Mini 4B Realtime BF16 GGUF","precision":"bf16"}],"docs":["docs/models/voxtral_realtime.md","docs/asr.md","docs/gguf.md"],"variants":["Voxtral Mini 4B Realtime"],"inputs":[{"type":"audio","label":"Speech stream","required":true},{"type":"control","label":"Transcription delay","required":false}],"detailStatus":"Causal acoustic encoder, audio/text addition and delay-conditioning branches audited.","usageDoc":"docs/models/voxtral_realtime.md"},{"id":"samsone","family":"samsone","name":"SAMSOne","task":"speech-recognition-audio-understanding","tasks":["asr"],"summary":"Whisper encodes audio; a pooled residual MLP maps it into a compact SmolLM2-derived decoder. Prompted generation can describe audio or answer audio questions, not only transcribe speech.","routes":[{"name":"Audio question answering / transcription","nodes":[{"id":"audio","kind":"input","label":"Audio"},{"id":"text","kind":"input","label":"Question / instruction","optional":true},{"id":"mel","component":"frontend-logmel"},{"id":"encoder","component":"encoder-whisper"},{"id":"adapter","component":"encoder-pooled-mlp-adaptor"},{"id":"tokenizer","component":"frontend-bpe","detail":"SmolLM2-derived tokenizer with SAMSONE's pruned-vocabulary remapping."},{"id":"ar","component":"ar-smollm2"},{"id":"out","kind":"output","label":"Text answer"}],"edges":[{"from":"audio","to":"mel"},{"from":"mel","to":"encoder"},{"from":"encoder","to":"adapter"},{"from":"adapter","to":"ar"},{"from":"text","to":"tokenizer","optional":true},{"from":"tokenizer","to":"ar"},{"from":"ar","to":"out"}]}],"sources":["https://github.com/SamsungLabs/samsone","https://github.com/SamsungLabs/samsone/blob/main/configs/Samsone134M.yaml"],"packages":[{"id":"samsone_99m_bf16","display_name":"SAMSONE 99M GGUF BF16","precision":"bf16"},{"id":"samsone_99m_q8_0","display_name":"SAMSONE 99M GGUF Q8_0","precision":"q8_0"},{"id":"samsone_134m_bf16","display_name":"SAMSONE 134M GGUF BF16","precision":"bf16"},{"id":"samsone_134m_q8_0","display_name":"SAMSONE 134M GGUF Q8_0","precision":"q8_0"},{"id":"samsone_356m_bf16","display_name":"SAMSONE 356M GGUF BF16","precision":"bf16"},{"id":"samsone_356m_q8_0","display_name":"SAMSONE 356M GGUF Q8_0","precision":"q8_0"}],"docs":["docs/models/samsone.md"],"variants":["SAMSONE 99M","SAMSONE 134M","SAMSONE 356M"],"inputs":[{"type":"audio","label":"Audio","required":true},{"type":"text","label":"Question / instruction","required":false}],"detailStatus":"Whisper pooling, residual projector and prompt branches audited.","usageDoc":"docs/models/samsone.md"},{"id":"canary_asr","family":"canary_asr","name":"Canary 180M Flash","task":"speech-recognition-audio-understanding","tasks":["asr"],"summary":"FastConformer encodes audio; a cross-attention AR decoder generates text. Source/target language and punctuation controls become decoder prompt tokens, not acoustic features.","routes":[{"name":"Transcription / translation","nodes":[{"id":"audio","kind":"input","label":"Speech audio"},{"id":"control","kind":"input","label":"Language / task / punctuation"},{"id":"mel","component":"frontend-logmel"},{"id":"encoder","component":"encoder-fastconformer","detail":"17 encoder layers; factor-eight subsampling."},{"id":"decoder","component":"ar-nemo-cross-attention","detail":"4 decoder layers; language-specific SentencePiece vocabulary and control tokens."},{"id":"out","kind":"output","label":"Transcript / translated text"}],"edges":[{"from":"audio","to":"mel"},{"from":"mel","to":"encoder"},{"from":"encoder","to":"decoder","label":"Audio memory"},{"from":"control","to":"decoder","label":"Prompt token IDs"},{"from":"decoder","to":"out"}]}],"sources":["https://huggingface.co/nvidia/canary-180m-flash"],"packages":[{"id":"canary_180m_flash_f32","display_name":"Canary 180M Flash GGUF F32","precision":"f32"},{"id":"canary_180m_flash_q8_0","display_name":"Canary 180M Flash GGUF Q8_0","precision":"q8_0"}],"docs":["docs/models/canary_asr.md"],"variants":["Canary 180M Flash"],"inputs":[{"type":"audio","label":"Speech","required":true},{"type":"control","label":"Source / target language and punctuation","required":false}],"detailStatus":"Audio and decoder-prompt branches audited.","usageDoc":"docs/models/canary_asr.md"},{"id":"cohere_asr","family":"cohere_asr","name":"Cohere Transcribe","task":"speech-recognition-audio-understanding","tasks":["asr"],"summary":"A Conformer encoder and cross-attention Transformer decoder transcribe speech. Language and punctuation select decoder control tokens; this route does not imply translation or speaker diarization.","routes":[{"name":"Multilingual transcription","nodes":[{"id":"audio","kind":"input","label":"Speech audio"},{"id":"control","kind":"input","label":"Language / punctuation"},{"id":"mel","component":"frontend-logmel"},{"id":"encoder","component":"encoder-cohere-conformer"},{"id":"decoder","component":"ar-nemo-cross-attention"},{"id":"out","kind":"output","label":"Transcript"}],"edges":[{"from":"audio","to":"mel"},{"from":"mel","to":"encoder"},{"from":"encoder","to":"decoder","label":"Audio memory"},{"from":"control","to":"decoder","label":"Prompt token IDs"},{"from":"decoder","to":"out"}]}],"sources":["https://huggingface.co/CohereLabs/cohere-transcribe-03-2026"],"packages":[{"id":"cohere_transcribe_bf16","display_name":"Cohere Transcribe GGUF BF16","precision":"bf16"},{"id":"cohere_transcribe_q8_0","display_name":"Cohere Transcribe GGUF Q8_0","precision":"q8_0"},{"id":"cohere_transcribe_q4_0","display_name":"Cohere Transcribe GGUF Q4_0","precision":"q4_0"}],"docs":["docs/models/cohere_asr.md"],"variants":["Cohere Transcribe"],"inputs":[{"type":"audio","label":"Speech","required":true},{"type":"control","label":"Language / punctuation","required":false}],"detailStatus":"Audio and decoder-prompt branches audited.","usageDoc":"docs/models/cohere_asr.md"},{"id":"hviske_asr","family":"hviske_asr","name":"Hviske ASR","task":"speech-recognition-audio-understanding","tasks":["asr"],"summary":"Danish encoder-decoder ASR: log-mel features pass through convolutional subsampling and a relative-position Conformer, followed by autoregressive text decoding. Punctuation is a decoder control token.","routes":[{"name":"Danish speech recognition","nodes":[{"id":"audio","kind":"input","label":"Speech audio"},{"id":"control","kind":"input","label":"Punctuation control","optional":true},{"id":"mel","component":"frontend-logmel"},{"id":"encoder","component":"encoder-hviske-conformer"},{"id":"decoder","component":"ar-nemo-cross-attention"},{"id":"out","kind":"output","label":"Transcript"}],"edges":[{"from":"audio","to":"mel"},{"from":"mel","to":"encoder"},{"from":"encoder","to":"decoder"},{"from":"control","to":"decoder","label":"Control token","optional":true},{"from":"decoder","to":"out"}]}],"sources":["https://huggingface.co/syvai/hviske-v5.3"],"packages":[{"id":"hviske_asr_q8_0","display_name":"Hviske v5.3 Q8_0 GGUF","precision":"q8_0"},{"id":"hviske_asr_safetensors","display_name":"Hviske v5.3 Safetensors","precision":"native"}],"docs":["docs/asr.md","docs/gguf.md"],"variants":["Hviske v5.3"],"inputs":[{"type":"audio","label":"Speech","required":true},{"type":"control","label":"Punctuation control","required":false}],"detailStatus":"Conformer encoder and cross-attention decoder route audited.","usageDoc":"docs/asr.md#hviske-asr"},{"id":"moonshine_asr","family":"moonshine_asr","name":"Moonshine Streaming ASR","task":"speech-recognition-audio-understanding","tasks":["asr"],"summary":"Moonshine Streaming uses normalized waveform frames, causal convolutions and a local-attention encoder without positional embeddings. An adapter adds positions before a separate autoregressive text decoder.","routes":[{"name":"Streaming-checkpoint speech recognition","nodes":[{"id":"audio","kind":"input","label":"Speech audio"},{"id":"frontend","component":"frontend-moonshine-time-domain"},{"id":"encoder","component":"encoder-moonshine-windowed"},{"id":"adapter","component":"encoder-moonshine-position-adapter"},{"id":"decoder","component":"ar-moonshine-cross-attention"},{"id":"out","kind":"output","label":"Transcript"}],"edges":[{"from":"audio","to":"frontend"},{"from":"frontend","to":"encoder"},{"from":"encoder","to":"adapter"},{"from":"adapter","to":"decoder"},{"from":"decoder","to":"out"}]}],"sources":["https://huggingface.co/moonshine-ai/moonshine-streaming-tiny"],"packages":[{"id":"moonshine_streaming_tiny_q8_0","display_name":"Moonshine Streaming Tiny Q8_0 GGUF","precision":"q8_0"},{"id":"moonshine_streaming_small_q8_0","display_name":"Moonshine Streaming Small Q8_0 GGUF","precision":"q8_0"},{"id":"moonshine_streaming_medium_q8_0","display_name":"Moonshine Streaming Medium Q8_0 GGUF","precision":"q8_0"},{"id":"moonshine_streaming_tiny_safetensors","display_name":"Moonshine Streaming Tiny Safetensors","precision":"f32"},{"id":"moonshine_streaming_small_safetensors","display_name":"Moonshine Streaming Small Safetensors","precision":"f32"},{"id":"moonshine_streaming_medium_safetensors","display_name":"Moonshine Streaming Medium Safetensors","precision":"f32"}],"docs":["docs/asr.md","docs/models/moonshine_asr.md","docs/gguf.md"],"variants":["Moonshine Streaming Tiny","Moonshine Streaming Small","Moonshine Streaming Medium"],"inputs":[{"type":"audio","label":"Speech","required":true}],"detailStatus":"Time-domain frontend, sliding-window encoder and positional adapter audited.","usageDoc":"docs/models/moonshine_asr.md"},{"id":"nemotron_asr","family":"nemotron_asr","name":"Nemotron 3.5 ASR","task":"speech-recognition-audio-understanding","tasks":["asr"],"summary":"Cache-aware FastConformer and RNN-T recognize speech. In Nemotron 3.5, language conditioning is fused with encoded acoustic frames before the transducer, unlike Canary's decoder prompt tokens.","routes":[{"name":"Offline / cache-aware streaming ASR","nodes":[{"id":"audio","kind":"input","label":"Speech audio / live chunks"},{"id":"language","kind":"input","label":"Language ID / auto","optional":true},{"id":"mel","component":"frontend-logmel"},{"id":"encoder","component":"encoder-fastconformer","detail":"Cache-aware causal subsampling, self-attention and convolution state."},{"id":"prompt","component":"encoder-language-fusion"},{"id":"head","component":"head-rnn-t"},{"id":"out","kind":"output","label":"Transcript + token timing"}],"edges":[{"from":"audio","to":"mel"},{"from":"mel","to":"encoder"},{"from":"encoder","to":"prompt"},{"from":"language","to":"prompt","optional":true},{"from":"prompt","to":"head"},{"from":"head","to":"out"}]}],"sources":["https://huggingface.co/nvidia/nemotron-3.5-asr-streaming-0.6b"],"packages":[{"id":"nemotron_asr_q8_0","display_name":"Nemotron 3.5 ASR Streaming 0.6B Q8_0 GGUF","precision":"q8_0"},{"id":"nemotron_asr_f16","display_name":"Nemotron 3.5 ASR Streaming 0.6B F16 GGUF","precision":"f16"},{"id":"nemotron_asr_safetensors","display_name":"Nemotron 3.5 ASR Streaming 0.6B Safetensors","precision":"native"}],"docs":["docs/asr.md","docs/gguf.md"],"variants":["Nemotron 3.5 ASR Streaming 0.6B"],"inputs":[{"type":"audio","label":"Speech / live chunks","required":true},{"type":"control","label":"Language ID or automatic detection","required":false}],"detailStatus":"Core offline / streaming recognition route audited; speaker-tagging extension not expanded.","usageDoc":"docs/asr.md#nemotron-asr"},{"id":"parakeet_tdt","family":"parakeet_tdt","name":"Parakeet TDT","task":"speech-recognition-audio-understanding","tasks":["asr"],"summary":"FastConformer encodes acoustic frames. The TDT decoder jointly predicts text tokens and frame-advance durations; it does not use a separate large language model.","routes":[{"name":"Token-duration transcription","nodes":[{"id":"audio","kind":"input","label":"Speech audio"},{"id":"mel","component":"frontend-logmel"},{"id":"encoder","component":"encoder-fastconformer"},{"id":"head","component":"head-tdt"},{"id":"out","kind":"output","label":"Transcript + token timing"}],"edges":[{"from":"audio","to":"mel"},{"from":"mel","to":"encoder"},{"from":"encoder","to":"head"},{"from":"head","to":"out"}]}],"sources":["https://huggingface.co/nvidia/parakeet-tdt-0.6b-v3"],"packages":[{"id":"parakeet_tdt_q8_0","display_name":"Parakeet-TDT 0.6B v3 Q8_0 GGUF","precision":"q8_0"},{"id":"parakeet_tdt_f16","display_name":"Parakeet-TDT 0.6B v3 F16 GGUF","precision":"f16"},{"id":"orukeet_q8_0","display_name":"Orukeet r3 Q8_0 GGUF (Parakeet-TDT 0.6B v3 fine-tune)","precision":"q8_0"},{"id":"orukeet_f16","display_name":"Orukeet r3 F16 GGUF (Parakeet-TDT 0.6B v3 fine-tune)","precision":"f16"}],"docs":["docs/community_models/parakeet_tdt.md"],"variants":["Parakeet-TDT 0.6B v3","Orukeet r3 (Parakeet-TDT 0.6B v3 fine-tune)"],"inputs":[{"type":"audio","label":"Speech","required":true}],"detailStatus":"Acoustic encoder and token-duration decoding audited.","usageDoc":"docs/community_models/parakeet_tdt.md"},{"id":"gigaam_asr","family":"gigaam_asr","name":"GigaAM ASR","task":"speech-recognition-audio-understanding","tasks":["asr"],"summary":"A rotary Conformer encodes speech. The selected checkpoint uses either a CTC framewise classifier or an RNN-T predictor and joint network. Language is inferred from audio, not supplied as a decoder prompt.","routes":[{"name":"v3 CTC / v3 E2E CTC / multilingual CTC","nodes":[{"id":"audio","kind":"input","label":"Speech audio"},{"id":"mel","component":"frontend-logmel"},{"id":"encoder","component":"encoder-gigaam-conformer"},{"id":"head","component":"head-ctc"},{"id":"out","kind":"output","label":"Transcript"}],"edges":[{"from":"audio","to":"mel"},{"from":"mel","to":"encoder"},{"from":"encoder","to":"head"},{"from":"head","to":"out"}]},{"name":"v3 RNN-T / v3 E2E RNN-T","nodes":[{"id":"audio","kind":"input","label":"Speech audio"},{"id":"mel","component":"frontend-logmel"},{"id":"encoder","component":"encoder-gigaam-conformer"},{"id":"head","component":"head-rnn-t"},{"id":"out","kind":"output","label":"Transcript"}],"edges":[{"from":"audio","to":"mel"},{"from":"mel","to":"encoder"},{"from":"encoder","to":"head"},{"from":"head","to":"out"}]}],"sources":["https://huggingface.co/ai-sage/GigaAM-v3","https://huggingface.co/ai-sage/GigaAM-Multilingual"],"packages":[{"id":"gigaam_v3_ctc_f16","display_name":"GigaAM v3 CTC F16 GGUF","precision":"f16"},{"id":"gigaam_v3_rnnt_f16","display_name":"GigaAM v3 RNN-T F16 GGUF","precision":"f16"},{"id":"gigaam_v3_e2e_ctc_f16","display_name":"GigaAM v3 E2E CTC F16 GGUF","precision":"f16"},{"id":"gigaam_v3_e2e_rnnt_f16","display_name":"GigaAM v3 E2E RNN-T F16 GGUF","precision":"f16"},{"id":"gigaam_multilingual_ctc_f32","display_name":"GigaAM Multilingual CTC F32 GGUF","precision":"f32"},{"id":"gigaam_multilingual_ctc_f16","display_name":"GigaAM Multilingual CTC F16 GGUF","precision":"f16"},{"id":"gigaam_multilingual_large_ctc_f32","display_name":"GigaAM Multilingual Large CTC F32 GGUF","precision":"f32"},{"id":"gigaam_multilingual_large_ctc_f16","display_name":"GigaAM Multilingual Large CTC F16 GGUF","precision":"f16"}],"docs":["docs/models/gigaam_asr.md"],"variants":["GigaAM v3 CTC","GigaAM v3 RNN-T","GigaAM v3 E2E CTC","GigaAM v3 E2E RNN-T","GigaAM Multilingual CTC","GigaAM Multilingual Large CTC"],"inputs":[{"type":"audio","label":"Speech","required":true}],"detailStatus":"CTC and transducer routes audited; multilingual packages use CTC.","usageDoc":"docs/models/gigaam_asr.md"},{"id":"kroko_asr","family":"kroko_asr","name":"Kroko Community ASR","task":"speech-recognition-audio-understanding","tasks":["asr"],"summary":"Kroko combines a multi-rate Zipformer acoustic encoder with a transducer. Its token predictor uses two-token context and grouped convolution, not a recurrent LSTM state.","routes":[{"name":"Streaming transducer recognition","nodes":[{"id":"audio","kind":"input","label":"Speech audio"},{"id":"frontend","component":"frontend-kaldi-fbank"},{"id":"encoder","component":"encoder-zipformer"},{"id":"head","component":"head-stateless-transducer"},{"id":"out","kind":"output","label":"Transcript"}],"edges":[{"from":"audio","to":"frontend"},{"from":"frontend","to":"encoder"},{"from":"encoder","to":"head"},{"from":"head","to":"out"}]}],"sources":["https://huggingface.co/Banafo/Kroko-ASR"],"packages":[{"id":"kroko_asr_community_q8_0","display_name":"Kroko Community ASR Q8_0 GGUF","precision":"q8_0"}],"docs":["docs/asr.md","docs/community_models/kroko_asr.md","docs/gguf.md"],"variants":["Kroko Community ASR"],"inputs":[{"type":"audio","label":"Speech","required":true}],"detailStatus":"Filterbank, multi-rate Zipformer and stateless transducer route audited.","usageDoc":"docs/community_models/kroko_asr.md"},{"id":"granite5asr","family":"granite5asr","name":"Granite Speech 5.0","task":"speech-recognition-audio-understanding","tasks":["asr"],"summary":"A block-attention Conformer consumes stacked log-mel frames. Intermediate CTC probabilities are projected back into the hidden representation before final CTC decoding.","routes":[{"name":"TurboCTC speech recognition","nodes":[{"id":"audio","kind":"input","label":"Speech audio"},{"id":"mel","component":"frontend-logmel","detail":"Log-mel extraction with neighboring-frame stacking."},{"id":"encoder","component":"encoder-granite-conformer"},{"id":"head","component":"head-ctc"},{"id":"out","kind":"output","label":"Transcript"}],"edges":[{"from":"audio","to":"mel"},{"from":"mel","to":"encoder"},{"from":"encoder","to":"head"},{"from":"head","to":"out"}]}],"sources":["https://huggingface.co/ibm-granite/granite-speech-5.0-470m-turboctc"],"packages":[{"id":"granite5asr_q8_0","display_name":"Granite Speech 5.0 470M TurboCTC Q8_0 GGUF","precision":"q8_0"},{"id":"granite5asr_safetensors","display_name":"Granite Speech 5.0 470M TurboCTC Native Safetensors","precision":"bfloat16"}],"docs":["docs/asr.md","docs/gguf.md"],"variants":["Granite Speech 5.0 470M TurboCTC"],"inputs":[{"type":"audio","label":"Speech","required":true}],"detailStatus":"Stacked mel features, intermediate conditioning and CTC route audited.","usageDoc":"docs/community_models/granite5asr.md"},{"id":"citrinet_asr","family":"citrinet_asr","name":"Citrinet ASR","task":"speech-recognition-audio-understanding","tasks":["asr"],"summary":"Citrinet uses residual separable temporal convolutions and squeeze-and-excitation on log-mel features, followed by a CTC head. It has no AR text decoder.","routes":[{"name":"CTC speech recognition","nodes":[{"id":"audio","kind":"input","label":"Speech audio"},{"id":"mel","component":"frontend-logmel"},{"id":"encoder","component":"encoder-citrinet"},{"id":"head","component":"head-ctc"},{"id":"out","kind":"output","label":"Transcript"}],"edges":[{"from":"audio","to":"mel"},{"from":"mel","to":"encoder"},{"from":"encoder","to":"head"},{"from":"head","to":"out"}]}],"sources":["https://huggingface.co/nvidia/stt_en_citrinet_256_ls"],"packages":[{"id":"citrinet_asr_q8_0","display_name":"Citrinet ASR Q8_0 GGUF","precision":"q8_0"}],"docs":["docs/asr.md","docs/gguf.md"],"variants":["Citrinet ASR"],"inputs":[{"type":"audio","label":"Speech","required":true}],"detailStatus":"Convolutional encoder and CTC route audited.","usageDoc":"docs/asr.md#citrinet-asr"},{"id":"sense_asr","family":"sense_asr","name":"SenseVoice-Small","task":"speech-recognition-audio-understanding","tasks":["asr"],"summary":"SenseVoiceSmall combines SANM acoustic encoding with CTC classification. Language and normalization controls enter as learned query embeddings; generated labels can also identify emotion and audio events.","routes":[{"name":"Speech recognition and audio labels","nodes":[{"id":"audio","kind":"input","label":"Speech / audio"},{"id":"control","kind":"input","label":"Language / normalization","optional":true},{"id":"features","component":"frontend-fbank-lfr"},{"id":"encoder","component":"encoder-sensevoice-sanm"},{"id":"head","component":"head-ctc"},{"id":"out","kind":"output","label":"Transcript + audio labels"}],"edges":[{"from":"audio","to":"features"},{"from":"features","to":"encoder"},{"from":"control","to":"encoder","label":"Query embeddings","optional":true},{"from":"encoder","to":"head"},{"from":"head","to":"out"}]}],"sources":["https://huggingface.co/FunAudioLLM/SenseVoiceSmall"],"packages":[{"id":"sensevoice_small_q8","display_name":"SenseVoice-Small Q8 GGUF","precision":"q8_0"}],"docs":["docs/community_models/sense_asr.md"],"variants":["SenseVoice-Small"],"inputs":[{"type":"audio","label":"Speech / audio","required":true},{"type":"control","label":"Language / text normalization","required":false}],"detailStatus":"Filterbank and control-prefix branches audited.","usageDoc":"docs/community_models/sense_asr.md"},{"id":"niagara_asr","family":"niagara_asr","name":"Niagara ASR","task":"speech-recognition-audio-understanding","tasks":["asr"],"summary":"Niagara combines relative attention with state-space temporal filters and CTC decoding. Its Python frontend returns log-mel features despite describing them as MFCCs; the batch checkpoints do not imply the separate commercial streaming model.","routes":[{"name":"Batch speech recognition","nodes":[{"id":"audio","kind":"input","label":"Speech audio"},{"id":"mel","component":"frontend-logmel"},{"id":"encoder","component":"encoder-niagara-ssm-attention","detail":"Conv2D subsampling followed by feed-forward, attention and state-space layers."},{"id":"head","component":"head-ctc"},{"id":"out","kind":"output","label":"Transcript"}],"edges":[{"from":"audio","to":"mel"},{"from":"mel","to":"encoder"},{"from":"encoder","to":"head"},{"from":"head","to":"out"}]}],"sources":["https://huggingface.co/abr-ai/niagara-19m-batch.en","https://huggingface.co/abr-ai/niagara-19m-batch.en/blob/main/feature_extraction.py"],"packages":[{"id":"niagara_19m_f32","display_name":"Niagara 19M Batch English F32 GGUF","precision":"f32"},{"id":"niagara_38m_f32","display_name":"Niagara 38M Batch English F32 GGUF","precision":"f32"}],"docs":["docs/asr.md","docs/gguf.md"],"variants":["Niagara 19M Batch English","Niagara 38M Batch English"],"inputs":[{"type":"audio","label":"Speech","required":true}],"detailStatus":"Log-mel frontend and SSM-attention route audited.","usageDoc":"docs/asr.md#niagara-asr"},{"id":"auk","family":"auk","name":"AuK","task":"speech-recognition-audio-understanding","tasks":["tts","edit"],"summary":"Qwen2.5-Omni supplies contextual conditioning to a dual/single-stream flow Transformer. Source audio also supplies a VAE latent prefix. Base and AuK-Flash share this pipeline; Flash uses a distilled sampling schedule. Output is audio, not an ASR transcript.","routes":[{"name":"Instruction TTS","inputs":[{"type":"text","label":"Text + voice description","required":true},{"type":"duration","label":"Output duration","required":true}],"nodes":[{"id":"text","kind":"input","label":"Text + voice instruction"},{"id":"tok","component":"frontend-bpe"},{"id":"cond","component":"encoder-qwen2-5-omni"},{"id":"flow","component":"flow-conditional-transformer"},{"id":"vae","component":"codec-convolutional-vae"},{"id":"out","kind":"output","label":"24 kHz mono audio"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"cond"},{"from":"cond","to":"flow"},{"from":"flow","to":"vae","label":"Continuous latents"},{"from":"vae","to":"out"}]},{"name":"Cloning / editing / enhancement / separation","inputs":[{"type":"text","label":"Task instruction","required":true},{"type":"audio","label":"Reference or source audio","required":true}],"nodes":[{"id":"text","kind":"input","label":"Task instruction + optional target words"},{"id":"audio","kind":"input","label":"Reference / source audio"},{"id":"tok","component":"frontend-bpe"},{"id":"audioenc","component":"encoder-auk-audio"},{"id":"cond","component":"encoder-qwen2-5-omni"},{"id":"refvae","component":"encoder-auk-vae"},{"id":"flow","component":"flow-conditional-transformer"},{"id":"vae","component":"codec-convolutional-vae"},{"id":"out","kind":"output","label":"Generated / edited audio"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"cond"},{"from":"audio","to":"audioenc"},{"from":"audioenc","to":"cond","label":"Audio embeddings"},{"from":"audio","to":"refvae"},{"from":"refvae","to":"flow","label":"Reference latent prefix"},{"from":"cond","to":"flow","label":"Contextual instruction states"},{"from":"flow","to":"vae"},{"from":"vae","to":"out"}]}],"sources":["https://huggingface.co/tencent/AuK"],"packages":[{"id":"auk_base_f32","display_name":"AuK Base F32","precision":"f32"},{"id":"auk_base_f16","display_name":"AuK Base F16","precision":"f16"},{"id":"auk_base_q8_0","display_name":"AuK Base Q8_0","precision":"q8_0"},{"id":"auk_flash_f32","display_name":"AuK-Flash F32","precision":"f32"},{"id":"auk_flash_f16","display_name":"AuK-Flash F16","precision":"f16"},{"id":"auk_flash_q8_0","display_name":"AuK-Flash Q8_0","precision":"q8_0"},{"id":"auk_qwen_bf16","display_name":"AuK Qwen BF16","precision":"bf16"},{"id":"auk_qwen_q8_0","display_name":"AuK Qwen Q8_0","precision":"q8_0"},{"id":"auk_vae_f32","display_name":"AuK VAE F32","precision":"f32"}],"docs":["docs/community_models/auk.md"],"variants":["AuK Base","AuK-Flash"],"inputs":[{"type":"text","label":"Instruction / target text","required":true},{"type":"audio","label":"Reference or source audio","required":false}],"detailStatus":"Instruction, source-audio and continuous-latent conditioning audited for both variants.","usageDoc":"docs/community_models/auk.md"},{"id":"firered_audio","family":"firered_audio","name":"FireRedAudio","task":"speech-recognition-audio-understanding","tasks":["asr","tts","clone","design","edit"],"summary":"A shared hybrid Qwen3.5 backbone receives Whisper-style features for audio understanding, or RedAE patch embeddings for speech generation. Generation uses a patch flow DiT and RedAE waveform decoder; those stages are absent from ASR and audio QA.","routes":[{"name":"ASR / audio understanding","inputs":[{"type":"audio","label":"Input recording","required":true},{"type":"text","label":"Question / transcription prompt","required":false}],"nodes":[{"id":"audio","kind":"input","label":"Recording"},{"id":"text","kind":"input","label":"Question / prompt"},{"id":"enc","component":"encoder-firered-understanding"},{"id":"tok","component":"frontend-bpe"},{"id":"ar","component":"ar-qwen3-5"},{"id":"out","kind":"output","label":"Transcript / answer"}],"edges":[{"from":"audio","to":"enc"},{"from":"text","to":"tok"},{"from":"enc","to":"ar"},{"from":"tok","to":"ar"},{"from":"ar","to":"out"}]},{"name":"TTS cloning / semantic or acoustic editing","inputs":[{"type":"audio","label":"Reference voice or editing source","required":true},{"type":"text","label":"Text + reference transcript, or edit instruction","required":true}],"nodes":[{"id":"audio","kind":"input","label":"Reference / source audio"},{"id":"text","kind":"input","label":"Text / transcript / edit instruction"},{"id":"enc","component":"encoder-redae"},{"id":"patch","component":"encoder-firered-patch"},{"id":"tok","component":"frontend-bpe"},{"id":"ar","component":"ar-qwen3-5","detail":"Semantic editing first generates rewritten text, then audio patches."},{"id":"flow","component":"flow-dit-flow"},{"id":"codec","component":"codec-redae"},{"id":"out","kind":"output","label":"Generated speech"}],"edges":[{"from":"audio","to":"enc"},{"from":"enc","to":"patch"},{"from":"patch","to":"ar"},{"from":"text","to":"tok"},{"from":"tok","to":"ar"},{"from":"ar","to":"flow","label":"Patch conditioning"},{"from":"flow","to":"codec"},{"from":"codec","to":"out"}]},{"name":"Voice design","inputs":[{"type":"text","label":"Speech text + voice description","required":true}],"nodes":[{"id":"text","kind":"input","label":"Text + voice description"},{"id":"tok","component":"frontend-bpe"},{"id":"ar","component":"ar-qwen3-5"},{"id":"flow","component":"flow-dit-flow"},{"id":"codec","component":"codec-redae"},{"id":"out","kind":"output","label":"Generated speech"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"ar"},{"from":"ar","to":"flow"},{"from":"flow","to":"codec"},{"from":"codec","to":"out"}]}],"sources":["https://huggingface.co/FireRedTeam/FireRedAudio"],"packages":[{"id":"firered_audio_orig","display_name":"FireRedAudio Orig GGUF","precision":"orig"},{"id":"firered_audio_q8_0","display_name":"FireRedAudio Q8_0 GGUF","precision":"q8_0"}],"docs":["docs/models/firered_audio.md","docs/gguf.md"],"variants":["FireRedAudio"],"inputs":[{"type":"audio","label":"Input audio for understanding or editing","required":false},{"type":"text","label":"Text / question / instruction","required":false}],"detailStatus":"Decoupled understanding and generation pathways audited.","usageDoc":"docs/models/firered_audio.md"},{"id":"personaplex","family":"personaplex","name":"PersonaPlex","task":"speech-recognition-audio-understanding","tasks":["s2s"],"summary":"PersonaPlex uses a Moshi temporal AR Transformer and a within-frame depth Transformer, with Mimi encoding incoming speech and decoding generated codes. Text predictions participate internally; the current audio.cpp session exposes the speech output, not a transcript.","routes":[{"name":"Conversation with packaged voice","nodes":[{"id":"speech","kind":"input","label":"User speech"},{"id":"prompt","kind":"input","label":"Optional persona text","optional":true},{"id":"voice","kind":"input","label":"Packaged voice ID"},{"id":"enc","component":"encoder-mimi"},{"id":"text","component":"frontend-personaplex-text"},{"id":"preset","component":"encoder-personaplex-prompt"},{"id":"ar","component":"ar-moshi-temporal"},{"id":"depth","component":"ar-moshi-depformer"},{"id":"codec","component":"codec-mimi"},{"id":"out","kind":"output","label":"Assistant speech"}],"edges":[{"from":"speech","to":"enc"},{"from":"prompt","to":"text","optional":true},{"from":"voice","to":"preset"},{"from":"text","to":"ar","optional":true},{"from":"preset","to":"ar"},{"from":"enc","to":"ar","label":"User audio codes"},{"from":"ar","to":"depth","label":"Hidden state + text token"},{"from":"depth","to":"codec","label":"Assistant audio codes"},{"from":"codec","to":"out"}]},{"name":"Conversation with reference voice","nodes":[{"id":"speech","kind":"input","label":"User speech"},{"id":"ref","kind":"input","label":"Reference voice"},{"id":"prompt","kind":"input","label":"Optional persona text","optional":true},{"id":"enc","component":"encoder-mimi"},{"id":"refenc","component":"encoder-mimi","detail":"Reference speech is encoded and replayed as a voice-conditioning prefix, not used as user conversation input."},{"id":"text","component":"frontend-personaplex-text"},{"id":"ar","component":"ar-moshi-temporal"},{"id":"depth","component":"ar-moshi-depformer"},{"id":"codec","component":"codec-mimi"},{"id":"out","kind":"output","label":"Assistant speech"}],"edges":[{"from":"speech","to":"enc"},{"from":"ref","to":"refenc"},{"from":"prompt","to":"text","optional":true},{"from":"text","to":"ar","optional":true},{"from":"refenc","to":"ar","label":"Voice prompt prefix"},{"from":"enc","to":"ar","label":"User audio codes"},{"from":"ar","to":"depth","label":"Hidden state + text token"},{"from":"depth","to":"codec","label":"Assistant audio codes"},{"from":"codec","to":"out"}]}],"sources":["https://huggingface.co/nvidia/personaplex-7b-v1","https://github.com/NVIDIA/personaplex"],"packages":[{"id":"personaplex_7b_v1_q4_k","display_name":"PersonaPlex 7B v1 Q4_K GGUF","precision":"q4_k"},{"id":"personaplex_7b_v1_q8_0","display_name":"PersonaPlex 7B v1 Q8_0 GGUF","precision":"q8_0"}],"docs":["docs/models/personaplex.md","docs/tts.md","docs/gguf.md"],"variants":["PersonaPlex 7B v1"],"inputs":[{"type":"audio","label":"User speech","required":true},{"type":"text","label":"Persona / system prompt","required":false},{"type":"audio","label":"Reference voice or packaged voice prompt","required":false}],"detailStatus":"System-prompt tokenization, packaged versus raw voice prompts, delayed audio streams, temporal/depth AR blocks and quantized Mimi paths audited. Distinct from PocketTTS's continuous Mimi latents.","usageDoc":"docs/models/personaplex.md"},{"id":"yue2","family":"yue2","name":"YuE2","task":"music-sound-video-generation","tasks":["music"],"summary":"YuE2's Mixture-of-Transformers has distinct AR and NAR parameters. AR optionally plans an ABC score and generates semantic music tokens. NAR flow uses the AR prefix keys/values to generate continuous acoustic latents; an Oobleck-style decoder reconstructs stereo music. Mixture here is not a routed mixture of experts.","routes":[{"name":"Score / semantic planning to song","nodes":[{"id":"text","kind":"input","label":"Style + lyrics"},{"id":"abc","kind":"input","label":"Optional supplied ABC","optional":true},{"id":"tok","component":"frontend-bpe","detail":"YuE2 text/ABC vocabulary and structural markers."},{"id":"plan","component":"ar-mixture-of-transformers","detail":"Optional ABC planning; bypassed when planning is off or a score is supplied."},{"id":"semantic","component":"ar-mixture-of-transformers","detail":"Same AR branch predicts discrete semantic music tokens."},{"id":"flow","component":"flow-mixture-of-transformers"},{"id":"codec","component":"codec-oobleck","detail":"YuE2 listening decoder; 48 kHz stereo."},{"id":"score","kind":"output","label":"Optional ABC score"},{"id":"out","kind":"output","label":"Stereo song"}],"edges":[{"from":"text","to":"tok"},{"from":"abc","to":"tok","optional":true},{"from":"tok","to":"plan","optional":true},{"from":"tok","to":"semantic"},{"from":"plan","to":"semantic","optional":true,"label":"Planned ABC"},{"from":"plan","to":"score","optional":true},{"from":"semantic","to":"flow","label":"Conditioned AR prefix K/V"},{"from":"flow","to":"codec","label":"Continuous latents"},{"from":"codec","to":"out"}]}],"sources":["https://github.com/multimodal-art-projection/YuE/blob/main/src/yue2/modeling_yue2.py"],"packages":[{"id":"yue2_main_q8_0","display_name":"Yue2 3B Main Q8_0","precision":"q8_0"},{"id":"yue2_main_bf16","display_name":"Yue2 3B Main BF16","precision":"bf16"},{"id":"yue2_main_q4_0","display_name":"Yue2 3B Main Q4_0","precision":"q4_0"},{"id":"yue2_vae_f16","display_name":"Yue2 VAE F16","precision":"f16"},{"id":"yue2_vae_f32","display_name":"Yue2 VAE F32","precision":"f32"}],"docs":["docs/models/yue2.md","docs/music_generation.md","docs/gguf.md"],"variants":["YuE2 3B"],"inputs":[{"type":"text","label":"Style + lyrics (lyrics may be empty)","required":true},{"type":"text","label":"Optional ABC score","required":false}],"detailStatus":"Symbolic planning, semantic AR and coupled NAR attention audited.","usageDoc":"docs/models/yue2.md"},{"id":"ace_step","family":"ace_step","name":"ACE-Step","task":"music-sound-video-generation","tasks":["music","edit"],"summary":"A Qwen3 AR planner can produce metadata and music codes. A separate Qwen3 text encoder plus lyric/timbre Transformers condition a cross-attention flow DiT. FSQ cover encoding replaces the planner for covers. A continuous convolutional VAE reconstructs stereo audio. XL separates condition-encoder and DiT widths and adds CLS timbre pooling.","routes":[{"name":"Text to music","inputs":[{"type":"text","label":"Description + optional lyrics","required":true}],"nodes":[{"id":"text","kind":"input","label":"Description + lyrics"},{"id":"tok","component":"frontend-bpe"},{"id":"planner","component":"ar-qwen3","detail":"Optional AR planner: metadata and 5 Hz audio codes; supplied codes bypass planning."},{"id":"textenc","component":"encoder-ace-qwen3"},{"id":"condition","component":"encoder-ace-condition"},{"id":"detok","component":"encoder-ace-detokenizer"},{"id":"flow","component":"flow-ace-dit"},{"id":"vae","component":"codec-ace-vae"},{"id":"out","kind":"output","label":"Stereo music"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"planner","optional":true},{"from":"tok","to":"textenc"},{"from":"planner","to":"textenc","optional":true,"label":"Metadata"},{"from":"textenc","to":"condition"},{"from":"planner","to":"detok","optional":true,"label":"Audio codes"},{"from":"detok","to":"flow","optional":true,"label":"Latent hints"},{"from":"condition","to":"flow"},{"from":"flow","to":"vae"},{"from":"vae","to":"out"}]},{"name":"Cover with FSQ","inputs":[{"type":"audio","label":"Source recording","required":true},{"type":"text","label":"Description + optional lyrics","required":true}],"nodes":[{"id":"audio","kind":"input","label":"Source audio"},{"id":"text","kind":"input","label":"Description + lyrics"},{"id":"tok","component":"frontend-bpe"},{"id":"textenc","component":"encoder-ace-qwen3"},{"id":"enc","component":"encoder-ace-vae"},{"id":"fsq","component":"encoder-ace-cover"},{"id":"detok","component":"encoder-ace-detokenizer"},{"id":"condition","component":"encoder-ace-condition"},{"id":"flow","component":"flow-ace-dit"},{"id":"vae","component":"codec-ace-vae"},{"id":"out","kind":"output","label":"Cover audio"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"textenc"},{"from":"textenc","to":"condition"},{"from":"audio","to":"enc"},{"from":"enc","to":"fsq"},{"from":"fsq","to":"detok"},{"from":"enc","to":"condition","label":"Reference timbre"},{"from":"detok","to":"flow","label":"Cover hints"},{"from":"condition","to":"flow"},{"from":"flow","to":"vae"},{"from":"vae","to":"out"}]},{"name":"Repaint / extract / cover-nofsq / complete / lego","inputs":[{"type":"audio","label":"Source audio; optional only for complete","required":false},{"type":"text","label":"Description / edit instruction + lyrics","required":true}],"nodes":[{"id":"audio","kind":"input","label":"Source recording"},{"id":"text","kind":"input","label":"Instruction + lyrics"},{"id":"tok","component":"frontend-bpe"},{"id":"planner","component":"ar-qwen3","detail":"Used by complete and lego; bypassed by repaint, extract and cover-nofsq."},{"id":"textenc","component":"encoder-ace-qwen3"},{"id":"enc","component":"encoder-ace-vae"},{"id":"detok","component":"encoder-ace-detokenizer"},{"id":"condition","component":"encoder-ace-condition"},{"id":"flow","component":"flow-ace-dit","detail":"Source latent context and route-specific masks preserve or edit the requested region."},{"id":"vae","component":"codec-ace-vae"},{"id":"out","kind":"output","label":"Edited / completed music"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"planner","optional":true},{"from":"tok","to":"textenc"},{"from":"textenc","to":"condition"},{"from":"audio","to":"enc"},{"from":"enc","to":"condition","label":"Timbre"},{"from":"enc","to":"flow","label":"Source latent / mask"},{"from":"planner","to":"detok","optional":true},{"from":"detok","to":"flow","optional":true},{"from":"condition","to":"flow"},{"from":"flow","to":"vae"},{"from":"vae","to":"out"}]}],"sources":["https://github.com/ace-step/ACE-Step-1.5"],"packages":[{"id":"ace_step_turbo_q8_0","display_name":"ACE-Step 1.5 Turbo Q8_0 GGUF","precision":"q8_0"},{"id":"ace_step_turbo_bf16","display_name":"ACE-Step 1.5 Turbo BF16 GGUF","precision":"bf16"},{"id":"ace_step_base_q8_0","display_name":"ACE-Step 1.5 Base Q8_0 GGUF","precision":"q8_0"},{"id":"ace_step_base_bf16","display_name":"ACE-Step 1.5 Base BF16 GGUF","precision":"bf16"},{"id":"ace_step_xl_turbo_bf16","display_name":"ACE-Step 1.5 XL Turbo BF16 GGUF","precision":"bf16"},{"id":"ace_step_xl_sft_bf16","display_name":"ACE-Step 1.5 XL SFT BF16 GGUF","precision":"bf16"},{"id":"ace_step_xl_turbo_q8dit","display_name":"ACE-Step 1.5 XL Turbo Q8_0 DiT GGUF (BF16 planner)","precision":"q8_0"},{"id":"ace_step_xl_sft_q8dit","display_name":"ACE-Step 1.5 XL SFT Q8_0 DiT GGUF (BF16 planner)","precision":"q8_0"}],"docs":["docs/models/ace_step.md","docs/music_generation.md","docs/gguf.md"],"variants":["1.5 Base","1.5 Turbo","1.5 XL SFT","1.5 XL Turbo"],"inputs":[{"type":"text","label":"Music description + optional lyrics","required":true},{"type":"audio","label":"Source / timbre reference, route-dependent","required":false}],"detailStatus":"Planner, text/lyric/timbre branches, FSQ cover and standard/XL conditioning audited.","usageDoc":"docs/models/ace_step.md"},{"id":"heartmula","family":"heartmula","name":"HeartMuLa","task":"music-sound-video-generation","tasks":["music"],"summary":"HeartMuLa generates audio codes with a Llama-style temporal AR backbone and a smaller within-frame AR decoder. HeartCodec reconstructs continuous latents using flow matching, then decodes stereo audio with a convolutional scalar decoder.","routes":[{"name":"Lyrics and tags to song","nodes":[{"id":"lyrics","kind":"input","label":"Lyrics + music tags"},{"id":"prompt","component":"frontend-heartmula-text"},{"id":"ar","component":"ar-heartmula-temporal"},{"id":"depth","component":"ar-heartmula-depth"},{"id":"codec","component":"codec-heartcodec"},{"id":"out","kind":"output","label":"Stereo song"}],"edges":[{"from":"lyrics","to":"prompt"},{"from":"prompt","to":"ar"},{"from":"ar","to":"depth","label":"Hidden state + first code"},{"from":"depth","to":"codec","label":"Complete codebook frames"},{"from":"codec","to":"out"}]}],"sources":["https://github.com/HeartMuLa/heartlib","https://huggingface.co/HeartMuLa/HeartMuLa-oss-3B"],"packages":[{"id":"heartmula_q8_0","display_name":"HeartMuLa Q8_0 GGUF","precision":"q8_0"},{"id":"heartmula_f16","display_name":"HeartMuLa F16 GGUF","precision":"f16"}],"docs":["docs/music_generation.md","docs/gguf.md"],"variants":["HeartMuLa"],"inputs":[{"type":"text","label":"Lyrics","required":true},{"type":"text","label":"Music tags","required":true}],"detailStatus":"Temporal versus codebook autoregression and HeartCodec reconstruction separated. The current prompt uses a zero continuous-conditioning vector; no reference-audio encoder is active.","usageDoc":"docs/music_generation.md#heartmula"},{"id":"minimax_music3","family":"minimax_music3","name":"MiniMax Music 3","task":"music-sound-video-generation","tasks":["music"],"summary":"Qwen3 predicts semantic music codes across time; a local AR decoder predicts the remaining codebooks within each frame. Their hidden states are fused to condition flow synthesis, rather than decoding the sampled tokens directly. The waveform decoder produces stereo music.","routes":[{"name":"Lyrics-conditioned song generation","nodes":[{"id":"text","kind":"input","label":"Music description + lyrics"},{"id":"tok","component":"frontend-bpe"},{"id":"global","component":"ar-qwen3","detail":"Qwen3-8B global music model; first codebook across frames."},{"id":"depth","component":"ar-local-depth-decoder"},{"id":"fusion","component":"encoder-music3-hidden-fusion"},{"id":"flow","component":"flow-music3"},{"id":"codec","component":"codec-flow-vae"},{"id":"out","kind":"output","label":"Stereo music"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"global"},{"from":"global","to":"depth","label":"Hidden state + semantic code"},{"from":"global","to":"fusion","label":"Global hidden state"},{"from":"depth","to":"fusion","label":"Local hidden states"},{"from":"fusion","to":"flow"},{"from":"flow","to":"codec","label":"Continuous latents"},{"from":"codec","to":"out"}]}],"sources":["https://huggingface.co/MiniMaxAI/MiniMax-Music3"],"packages":[{"id":"minimax_music3_q4_0","display_name":"MiniMax Music 3 Q4_0 GGUF","precision":"q4_0"},{"id":"minimax_music3_q8_0","display_name":"MiniMax Music 3 Q8_0 GGUF","precision":"q8_0"},{"id":"minimax_music3_bf16","display_name":"MiniMax Music 3 BF16 GGUF","precision":"bf16"}],"docs":["docs/music_generation.md","docs/gguf.md"],"variants":["MiniMax Music 3"],"inputs":[{"type":"text","label":"Music description","required":true},{"type":"text","label":"Lyrics","required":true}],"detailStatus":"Global/local hidden-state synthesis and text inputs audited.","usageDoc":"docs/community_models/minimax_music3.md"},{"id":"midashenglm_gen","family":"midashenglm_gen","name":"MiDashengLM-Gen","task":"music-sound-video-generation","tasks":["music"],"summary":"Qwen3 conditions a flow model that generates continuous audio patches. Each generated patch is projected back into the autoregressive sequence. A ConvNeXt / ISTFT decoder reconstructs speech, music or sound effects; this route does not accept a reference recording.","routes":[{"name":"Text to audio","nodes":[{"id":"text","kind":"input","label":"Description / structured caption"},{"id":"tok","component":"frontend-bpe"},{"id":"ar","component":"ar-midasheng-qwen3"},{"id":"flow","component":"flow-midasheng-patch"},{"id":"codec","component":"codec-midasheng-audio-tokenizer"},{"id":"out","kind":"output","label":"16 kHz mono audio"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"ar"},{"from":"ar","to":"flow","label":"Hidden conditioning"},{"from":"flow","to":"codec","label":"Continuous patches"},{"from":"codec","to":"out"}]}],"sources":["https://huggingface.co/mispeech/midashenglm-gen"],"packages":[{"id":"midashenglm_gen_f32","display_name":"MiDashengLM-Gen F32 GGUF","precision":"f32"},{"id":"midashenglm_gen_q8_0","display_name":"MiDashengLM-Gen Q8_0 GGUF","precision":"q8_0"}],"docs":["docs/models/midashenglm_gen.md","docs/gguf.md"],"variants":["MiDashengLM-Gen"],"inputs":[{"type":"text","label":"Audio description / structured caption","required":true}],"detailStatus":"Text conditioning, continuous-patch feedback and waveform synthesis audited.","usageDoc":"docs/models/midashenglm_gen.md"},{"id":"stable_audio","family":"stable_audio","name":"Stable Audio 3","task":"music-sound-video-generation","tasks":["music","sfx","edit"],"summary":"Stable Audio uses text-conditioned latent diffusion, not autoregressive audio-token generation. Foundation combines T5 with Oobleck; Stable Audio 3 combines T5Gemma with the SAME semantic-acoustic autoencoder. Audio initialization and masked editing activate an encoder branch.","routes":[{"name":"Foundation 1: text / optional audio initialization","nodes":[{"id":"text","kind":"input","label":"Prompt + optional negative prompt"},{"id":"time","kind":"input","label":"Start time + duration"},{"id":"audio","kind":"input","label":"Optional initialization audio","optional":true},{"id":"textenc","component":"encoder-t5"},{"id":"timing","component":"encoder-stable-timing","detail":"Foundation uses separate start and total-duration embeddings."},{"id":"enc","component":"encoder-oobleck"},{"id":"flow","component":"flow-stable-foundation"},{"id":"dec","component":"codec-oobleck"},{"id":"out","kind":"output","label":"Stereo audio"}],"edges":[{"from":"text","to":"textenc"},{"from":"time","to":"timing"},{"from":"audio","to":"enc","optional":true},{"from":"textenc","to":"flow"},{"from":"timing","to":"flow"},{"from":"enc","to":"flow","label":"Noised starting latent","optional":true},{"from":"flow","to":"dec"},{"from":"dec","to":"out"}]},{"name":"Stable Audio 3: music / SFX / masked editing","nodes":[{"id":"text","kind":"input","label":"Prompt + optional negative prompt"},{"id":"time","kind":"input","label":"Duration"},{"id":"audio","kind":"input","label":"Optional source audio","optional":true},{"id":"mask","kind":"input","label":"Optional edit regions","optional":true},{"id":"textenc","component":"encoder-t5-t5gemma"},{"id":"timing","component":"encoder-stable-timing","detail":"SA3 supplies total duration to both cross-attention and global conditioning."},{"id":"enc","component":"encoder-same"},{"id":"flow","component":"flow-stable3-rf-dit"},{"id":"dec","component":"codec-same"},{"id":"out","kind":"output","label":"Stereo audio"}],"edges":[{"from":"text","to":"textenc"},{"from":"time","to":"timing"},{"from":"audio","to":"enc","optional":true},{"from":"mask","to":"flow","label":"Keep / generate mask","optional":true},{"from":"textenc","to":"flow"},{"from":"timing","to":"flow"},{"from":"enc","to":"flow","label":"Initialization / masked latent context","optional":true},{"from":"flow","to":"dec"},{"from":"dec","to":"out"}]}],"sources":["https://github.com/Stability-AI/stable-audio-tools","https://huggingface.co/stabilityai/stable-audio-3-small-sfx","https://huggingface.co/stabilityai/stable-audio-3-medium"],"packages":[{"id":"stable_audio_3_medium_q8_0","display_name":"Stable Audio 3 Medium Q8_0 GGUF","precision":"q8_0"},{"id":"stable_audio_3_medium_f16","display_name":"Stable Audio 3 Medium F16 GGUF","precision":"f16"},{"id":"stable_audio_3_small_music_q8_0","display_name":"Stable Audio 3 Small Music Q8_0 GGUF","precision":"q8_0"},{"id":"stable_audio_3_small_music_f16","display_name":"Stable Audio 3 Small Music F16 GGUF","precision":"f16"},{"id":"stable_audio_3_small_sfx_q8_0","display_name":"Stable Audio 3 Small SFX Q8_0 GGUF","precision":"q8_0"},{"id":"stable_audio_3_small_sfx_f16","display_name":"Stable Audio 3 Small SFX F16 GGUF","precision":"f16"},{"id":"stable_audio_3_medium_safetensors","display_name":"Stable Audio 3 Medium Safetensors","precision":"native"},{"id":"stable_audio_3_small_music_safetensors","display_name":"Stable Audio 3 Small Music Safetensors","precision":"native"},{"id":"stable_audio_3_small_sfx_safetensors","display_name":"Stable Audio 3 Small SFX Safetensors","precision":"native"}],"docs":["docs/models/stable_audio.md","docs/music_generation.md","docs/gguf.md"],"variants":["Stable Audio 3 Medium","Stable Audio 3 Small Music","Stable Audio 3 Small SFX"],"inputs":[{"type":"text","label":"Sound / music prompt","required":true},{"type":"control","label":"Duration","required":false},{"type":"audio","label":"Optional initialization or edit audio","required":false}],"detailStatus":"Foundation and Stable Audio 3 text encoders, time conditioning, denoisers and autoencoders are separated. Foundation supports initialization but rejects the SA3 inpainting controls.","usageDoc":"docs/models/stable_audio.md"},{"id":"controlfoley","family":"controlfoley","name":"ControlFoley","task":"music-sound-video-generation","tasks":["sfx"],"summary":"ControlFoley uses separate semantic, visual-timing and reference-timbre encoders to condition a multimodal flow Transformer. A mel-latent VAE and BigVGAN synthesize the waveform. Text-only and video-conditioned routes activate different conditioning branches.","routes":[{"name":"T2A: text to sound","inputs":[{"type":"text","label":"Sound prompt","required":true}],"nodes":[{"id":"text","kind":"input","label":"Sound prompt"},{"id":"textenc","component":"encoder-openclip-text"},{"id":"flow","component":"flow-conditional-flow"},{"id":"vae","component":"codec-mel-latent-vae"},{"id":"vocoder","component":"codec-bigvgan"},{"id":"out","kind":"output","label":"Foley waveform"}],"edges":[{"from":"text","to":"textenc"},{"from":"textenc","to":"flow"},{"from":"flow","to":"vae"},{"from":"vae","to":"vocoder"},{"from":"vocoder","to":"out"}]},{"name":"V2A / TV2A / TC-V2A: video and optional text","inputs":[{"type":"video","label":"Video","required":true},{"type":"text","label":"Optional sound prompt","required":false}],"nodes":[{"id":"video","kind":"input","label":"Video frames"},{"id":"text","kind":"input","label":"Optional sound prompt","optional":true},{"id":"clip","component":"encoder-openclip-vision","detail":"mask_away_clip disables only this branch for text-controlled video generation."},{"id":"visual","component":"encoder-cavmae-visual"},{"id":"sync","component":"encoder-synchformer"},{"id":"textenc","component":"encoder-openclip-text"},{"id":"flow","component":"flow-conditional-flow"},{"id":"vae","component":"codec-mel-latent-vae"},{"id":"vocoder","component":"codec-bigvgan"},{"id":"out","kind":"output","label":"Synchronized Foley"}],"edges":[{"from":"video","to":"clip","optional":true},{"from":"video","to":"visual"},{"from":"video","to":"sync"},{"from":"text","to":"textenc","optional":true},{"from":"clip","to":"flow","optional":true},{"from":"visual","to":"flow"},{"from":"sync","to":"flow","label":"Timing"},{"from":"textenc","to":"flow","optional":true},{"from":"flow","to":"vae"},{"from":"vae","to":"vocoder"},{"from":"vocoder","to":"out"}]},{"name":"AC-V2A: reference sound and video","inputs":[{"type":"video","label":"Video","required":true},{"type":"audio","label":"Reference sound","required":true}],"nodes":[{"id":"video","kind":"input","label":"Video frames"},{"id":"audio","kind":"input","label":"Reference sound"},{"id":"clip","component":"encoder-openclip-vision"},{"id":"visual","component":"encoder-cavmae-visual"},{"id":"sync","component":"encoder-synchformer"},{"id":"clap","component":"encoder-clap-audio"},{"id":"timbre","component":"encoder-musicgen-style"},{"id":"flow","component":"flow-conditional-flow"},{"id":"vae","component":"codec-mel-latent-vae"},{"id":"vocoder","component":"codec-bigvgan"},{"id":"out","kind":"output","label":"Reference-conditioned Foley"}],"edges":[{"from":"video","to":"clip"},{"from":"video","to":"visual"},{"from":"video","to":"sync"},{"from":"audio","to":"clap"},{"from":"audio","to":"timbre"},{"from":"clip","to":"flow"},{"from":"visual","to":"flow"},{"from":"sync","to":"flow","label":"Timing"},{"from":"clap","to":"flow","label":"Audio semantics"},{"from":"timbre","to":"flow","label":"Global timbre"},{"from":"flow","to":"vae"},{"from":"vae","to":"vocoder"},{"from":"vocoder","to":"out"}]}],"sources":["https://huggingface.co/YJX-Xiaomi/ControlFoley","https://github.com/xiaomi-research/controlfoley"],"packages":[{"id":"controlfoley_large_44k_f32","display_name":"ControlFoley Large 44k F32 GGUF","precision":"f32"},{"id":"controlfoley_large_44k_q8_0","display_name":"ControlFoley Large 44k Q8_0 GGUF","precision":"q8_0"}],"docs":["docs/models/controlfoley.md","docs/gguf.md"],"variants":["ControlFoley Large 44k"],"inputs":[{"type":"text","label":"Sound prompt","required":false},{"type":"video","label":"Video conditioning","required":false},{"type":"audio","label":"Reference sound with video","required":false}],"detailStatus":"Five documented tasks covered by three structural routes: text-only, video with optional text, and reference audio with video. Disabling CLIP leaves CAV-MAE-ST and Synchformer active.","usageDoc":"docs/models/controlfoley.md"},{"id":"liveavatar","family":"liveavatar","name":"LiveAvatar","task":"music-sound-video-generation","tasks":["sfx"],"summary":"UMT5 encodes the prompt, XLS-R encodes driving audio, and a Wan VAE encodes the reference image. A distilled Wan S2V DiT generates video blocks with audio injection and retained history. The video VAE reconstructs frames; the driving audio is not synthesized by the model.","routes":[{"name":"Audio-driven avatar video","nodes":[{"id":"image","kind":"input","label":"Reference image"},{"id":"audio","kind":"input","label":"Driving audio"},{"id":"text","kind":"input","label":"Scene prompt"},{"id":"textenc","component":"encoder-umt5"},{"id":"audioenc","component":"encoder-liveavatar-xlsr"},{"id":"audioadapt","component":"encoder-wan-audio-condition"},{"id":"imageenc","component":"encoder-wan-vae"},{"id":"dit","component":"flow-wan-s2v-dit"},{"id":"vae","component":"codec-wan-video-vae"},{"id":"out","kind":"output","label":"Video frames + original audio"}],"edges":[{"from":"text","to":"textenc"},{"from":"audio","to":"audioenc"},{"from":"audioenc","to":"audioadapt"},{"from":"image","to":"imageenc"},{"from":"textenc","to":"dit"},{"from":"audioadapt","to":"dit"},{"from":"imageenc","to":"dit"},{"from":"dit","to":"vae"},{"from":"vae","to":"out"},{"from":"audio","to":"out","label":"Unchanged soundtrack"}]}],"sources":["https://github.com/Alibaba-Quark/LiveAvatar","https://huggingface.co/Wan-AI/Wan2.2-S2V-14B"],"packages":[{"id":"liveavatar_nvfp4_lora","display_name":"LiveAvatar NVFP4 LoRA GGUF","precision":"native"}],"docs":[],"variants":["LiveAvatar NVFP4 LoRA"],"inputs":[{"type":"image","label":"Reference image","required":true},{"type":"audio","label":"Driving audio","required":true},{"type":"text","label":"Scene / subject prompt","required":true}],"detailStatus":"UMT5, XLS-R, causal audio injection and Wan VAE branches audited.","usageDoc":"docs/community_models/liveavatar.md"},{"id":"minimax_h3","family":"minimax_h3","name":"MiniMax-H3","task":"music-sound-video-generation","tasks":["tts","music","sfx"],"summary":"Qwen3-VL's text backbone supplies prompt features to a joint audio/video flow DiT. The current port generates both latent streams even when video decode is disabled. A BigVGAN-based audio VAE reconstructs sound; an optional Transformer video VAE reconstructs frames. This entry does not claim upstream image-input support in audio.cpp.","routes":[{"name":"Text to audio / optional video","nodes":[{"id":"text","kind":"input","label":"Description"},{"id":"tok","component":"frontend-bpe"},{"id":"textenc","component":"encoder-qwen3-vl"},{"id":"dit","component":"flow-multimodal-dit"},{"id":"audio","component":"codec-h3-audio"},{"id":"video","component":"codec-h3-video","detail":"Only decoded when return_video=true."},{"id":"out","kind":"output","label":"Audio waveform"},{"id":"vout","kind":"output","label":"Optional video frames"}],"edges":[{"from":"text","to":"tok"},{"from":"tok","to":"textenc"},{"from":"textenc","to":"dit"},{"from":"dit","to":"audio","label":"Audio latents"},{"from":"dit","to":"video","optional":true,"label":"Video latents"},{"from":"audio","to":"out"},{"from":"video","to":"vout","optional":true}]}],"sources":["https://huggingface.co/MiniMaxAI/MiniMax-H3"],"packages":[{"id":"minimax_h3_q4_k","display_name":"MiniMax-H3 Q4_K GGUF","precision":"q4_k"},{"id":"minimax_h3_q4_k_int8_dit","display_name":"MiniMax-H3 Q4_K GGUF + INT8 ConvRot DiT","precision":"q4_k"}],"docs":["docs/community_models/minimax_h3.md","docs/reports/minimax_h3_performance.md"],"variants":["MiniMax-H3"],"inputs":[{"type":"text","label":"Audio / audiovisual description","required":true}],"detailStatus":"Current text-input route, joint latent solve and distinct decoders audited.","usageDoc":"docs/community_models/minimax_h3.md"},{"id":"sheetsage2","family":"sheetsage2","name":"SheetSage2","task":"music-transcription","tasks":["midi"],"summary":"SheetSage2 encodes normalized mel features with MERT2, then autoregressively predicts musical events using a cross-attention decoder. Event processing produces readable ABC notation; ABC formatting is not a neural backbone.","routes":[{"name":"Music to symbolic score","nodes":[{"id":"audio","kind":"input","label":"Music recording"},{"id":"mel","component":"frontend-sheetsage-mel"},{"id":"enc","component":"encoder-mert2"},{"id":"ar","component":"ar-sheetsage2-decoder"},{"id":"events","component":"head-abc-notation"},{"id":"out","kind":"output","label":"ABC score / musical events"}],"edges":[{"from":"audio","to":"mel"},{"from":"mel","to":"enc"},{"from":"enc","to":"ar","label":"Audio memory"},{"from":"ar","to":"events","label":"Symbolic token IDs"},{"from":"events","to":"out"}]}],"sources":["https://huggingface.co/m-a-p/SheetSage2"],"packages":[{"id":"sheetsage2_orig","display_name":"SheetSage2 Original-Dtype GGUF","precision":"orig"}],"docs":[],"variants":["SheetSage2"],"inputs":[{"type":"audio","label":"Music recording","required":true},{"type":"control","label":"Transcription task / window settings","required":false}],"detailStatus":"Mel frontend, ConvNeXt-style subsampling, rotary Conformer stack, layer mixing, post-norm AR decoder and symbolic formatting audited.","usageDoc":"docs/audio_tools.md#sheetsage2"},{"id":"muscriptor","family":"muscriptor","name":"MuScriptor","task":"music-transcription","tasks":["midi"],"summary":"MuScriptor projects log-mel frames into a prefix for a decoder-only AR Transformer. Instrument conditioning and preceding open-note tokens guide note-event generation; deterministic event decoding yields MIDI or JSON rather than audio.","routes":[{"name":"Music to MIDI events","nodes":[{"id":"audio","kind":"input","label":"Music recording"},{"id":"groups","kind":"input","label":"Optional instrument groups","optional":true},{"id":"mel","component":"frontend-logmel","detail":"Mono, resampled audio is split into fixed segments. Magnitude mel features use a natural logarithm."},{"id":"prefix","component":"encoder-audio-conditioning"},{"id":"ar","component":"ar-muscriptor-transformer"},{"id":"events","component":"head-midi-events"},{"id":"out","kind":"output","label":"MIDI / note-event JSON"}],"edges":[{"from":"audio","to":"mel"},{"from":"mel","to":"prefix"},{"from":"groups","to":"prefix","optional":true},{"from":"prefix","to":"ar"},{"from":"ar","to":"events"},{"from":"events","to":"out"}]}],"sources":["https://huggingface.co/MuScriptor/muscriptor-small","https://github.com/muscriptor/muscriptor"],"packages":[{"id":"muscriptor_small_f32","display_name":"MuScriptor Small F32 GGUF","precision":"f32"}],"docs":["docs/models/muscriptor.md","docs/gguf.md"],"variants":["MuScriptor Small"],"inputs":[{"type":"audio","label":"Music recording","required":true},{"type":"control","label":"Instrument groups","required":false}],"detailStatus":"Audio prefix projection, instrument/dataset embeddings, sinusoidal positions, pre-norm AR blocks and MIDI serialization audited. The audio.cpp streaming interface buffers input and returns a final result.","usageDoc":"docs/models/muscriptor.md"},{"id":"meanvc2","family":"meanvc2","name":"MeanVC2","task":"voice-conversion-audio-coding","tasks":["vc"],"summary":"MeanVC2 combines streaming WeNet bottleneck features with a WavLM/ECAPA speaker embedding. Learned timbre tokens provide pronunciation-dependent conditioning to a chunk-aware DiT mean-flow generator. Vocos synthesizes the generated mel frames.","routes":[{"name":"Reference-conditioned conversion","nodes":[{"id":"source","kind":"input","label":"Source speech"},{"id":"ref","kind":"input","label":"Target reference"},{"id":"fbank","component":"frontend-kaldi-fbank"},{"id":"content","component":"encoder-wenet-conformer"},{"id":"wavlm","component":"encoder-wavlm","detail":"Learned layer mixture and channel normalization precede ECAPA."},{"id":"speaker","component":"encoder-ecapa-features"},{"id":"timbre","component":"encoder-universal-timbre"},{"id":"flow","component":"flow-meanvc-dit"},{"id":"vocoder","component":"codec-vocos","detail":"Mel-conditioned 16 kHz Vocos."},{"id":"out","kind":"output","label":"Converted speech"}],"edges":[{"from":"source","to":"fbank"},{"from":"fbank","to":"content"},{"from":"ref","to":"wavlm"},{"from":"wavlm","to":"speaker"},{"from":"speaker","to":"timbre"},{"from":"content","to":"timbre","label":"Frame queries"},{"from":"timbre","to":"flow"},{"from":"speaker","to":"flow","label":"Global speaker condition"},{"from":"flow","to":"vocoder","label":"Mel frames"},{"from":"vocoder","to":"out"}]}],"sources":["https://huggingface.co/ASLP-lab/MeanVC2"],"packages":[{"id":"meanvc2_120ms_40ms_f32","display_name":"MeanVC2 120ms/40ms F32 GGUF","precision":"f32"},{"id":"meanvc2_120ms_40ms_q4_k","display_name":"MeanVC2 120ms/40ms Q4_K GGUF","precision":"q4_k"}],"docs":["docs/models/meanvc2.md","docs/audio_tools.md","docs/gguf.md"],"variants":["MeanVC2 120ms/40ms"],"inputs":[{"type":"audio","label":"Source speech","required":true},{"type":"audio","label":"Target speaker reference","required":true}],"detailStatus":"Content, speaker and timbre-token branches are separated; Conformer, DiT and Vocos blocks are expandable.","usageDoc":"docs/models/meanvc2.md"},{"id":"vevo2","family":"vevo2","name":"Vevo2","task":"voice-conversion-audio-coding","tasks":["tts","music","vc","edit","svc","s2s"],"summary":"Vevo2 separates content-style token generation from acoustic timbre synthesis. Style-preserved conversion directly tokenizes source audio; text and prosody routes use a Qwen2.5 AR model. Both feed a DiffLlama flow-matching mel generator conditioned on target-reference tokens and mel frames, then Vocos.","routes":[{"name":"Style-preserved VC / SVC (AR bypass)","inputs":[{"type":"audio","label":"Source recording","required":true},{"type":"audio","label":"Target timbre reference","required":true}],"nodes":[{"id":"source","kind":"input","label":"Source speech / singing"},{"id":"ref","kind":"input","label":"Target timbre reference"},{"id":"source_tokens","component":"encoder-coco-content-style"},{"id":"ref_tokens","component":"encoder-coco-content-style"},{"id":"mel","component":"frontend-logmel","detail":"Target-reference mel prompt for acoustic continuation."},{"id":"flow","component":"flow-diffllama"},{"id":"vocoder","component":"codec-vocos"},{"id":"out","kind":"output","label":"Converted speech / singing"}],"edges":[{"from":"source","to":"source_tokens"},{"from":"ref","to":"ref_tokens"},{"from":"ref","to":"mel"},{"from":"source_tokens","to":"flow"},{"from":"ref_tokens","to":"flow"},{"from":"mel","to":"flow"},{"from":"flow","to":"vocoder"},{"from":"vocoder","to":"out"}]},{"name":"Text / prosody generation and editing","inputs":[{"type":"text","label":"Target text / lyrics","required":true},{"type":"audio","label":"Target timbre reference","required":true},{"type":"audio","label":"Style reference + transcript","required":false},{"type":"audio","label":"Prosody source (route-dependent)","required":false}],"nodes":[{"id":"text","kind":"input","label":"Text / lyrics"},{"id":"style","kind":"input","label":"Style reference","optional":true},{"id":"prosody","kind":"input","label":"Prosody recording","optional":true},{"id":"ref","kind":"input","label":"Target timbre reference"},{"id":"bpe","component":"frontend-bpe"},{"id":"style_tokens","component":"encoder-coco-content-style"},{"id":"prosody_tokens","component":"encoder-coco-prosody"},{"id":"ar","component":"ar-qwen2","detail":"Qwen2.5-0.5B predicts content-style tokens from target text and optional reference/prosody tokens."},{"id":"ref_tokens","component":"encoder-coco-content-style"},{"id":"mel","component":"frontend-logmel"},{"id":"flow","component":"flow-diffllama"},{"id":"vocoder","component":"codec-vocos"},{"id":"out","kind":"output","label":"Generated speech / singing"}],"edges":[{"from":"text","to":"bpe"},{"from":"bpe","to":"ar"},{"from":"style","to":"style_tokens","optional":true},{"from":"style_tokens","to":"ar","optional":true},{"from":"prosody","to":"prosody_tokens","optional":true},{"from":"prosody_tokens","to":"ar","optional":true},{"from":"ref","to":"ref_tokens"},{"from":"ref","to":"mel"},{"from":"ar","to":"flow"},{"from":"ref_tokens","to":"flow"},{"from":"mel","to":"flow"},{"from":"flow","to":"vocoder"},{"from":"vocoder","to":"out"}]}],"sources":["https://huggingface.co/RMSnow/Vevo2"],"packages":[{"id":"vevo2_q8_0","display_name":"Vevo2 Q8_0 GGUF","precision":"q8_0"},{"id":"vevo2_f16","display_name":"Vevo2 F16 GGUF","precision":"f16"},{"id":"vevo2_orig","display_name":"Vevo2 Original-Dtype GGUF","precision":"orig"}],"docs":["docs/models/vevo2.md","docs/tts.md","docs/gguf.md"],"variants":["Vevo2"],"inputs":[{"type":"audio/text","label":"Content depends on route","required":true},{"type":"audio","label":"Target timbre reference","required":true},{"type":"audio","label":"Style / prosody reference","required":false}],"detailStatus":"AR and AR-bypass routes distinguished. CoCo prosody versus content-style tokenizer inputs and shared acoustic synthesis are expandable.","usageDoc":"docs/models/vevo2.md"},{"id":"seed_vc","family":"seed_vc","name":"Seed-VC","task":"voice-conversion-audio-coding","tasks":["vc","svc"],"summary":"Seed-VC converts a source recording using content features, target-speaker conditioning and mel flow matching. v1 checkpoints pair Whisper with BigVGAN or XLS-R with HiFT; the singing route adds pitch. The current audio.cpp v2 route directly uses HuBERT/ASTRAL tokens for CFM and does not run the upstream optional AR conversion stage.","routes":[{"name":"v2 VC: HuBERT / ASTRAL + BigVGAN","nodes":[{"id":"source","kind":"input","label":"Source recording"},{"id":"ref","kind":"input","label":"Target reference"},{"id":"content","component":"encoder-seed-hubert-astral"},{"id":"ref_content","component":"encoder-seed-hubert-astral","detail":"Reference tokens form the acoustic prompt prefix."},{"id":"length","component":"encoder-seed-length","detail":"Embed source/reference token IDs and resize each to its mel-frame length."},{"id":"fbank","component":"frontend-kaldi-fbank"},{"id":"speaker","component":"encoder-campplus"},{"id":"mel","component":"frontend-logmel"},{"id":"flow","component":"flow-seed-v2-cfm"},{"id":"vocoder","component":"codec-bigvgan"},{"id":"out","kind":"output","label":"22.05 kHz converted voice"}],"edges":[{"from":"source","to":"content"},{"from":"ref","to":"ref_content"},{"from":"content","to":"length"},{"from":"ref_content","to":"length"},{"from":"ref","to":"fbank"},{"from":"fbank","to":"speaker"},{"from":"ref","to":"mel"},{"from":"length","to":"flow"},{"from":"speaker","to":"flow","label":"Speaker style"},{"from":"mel","to":"flow","label":"Reference mel prompt"},{"from":"flow","to":"vocoder"},{"from":"vocoder","to":"out"}]},{"name":"v1 VC: Whisper + BigVGAN","nodes":[{"id":"source","kind":"input","label":"Source recording"},{"id":"ref","kind":"input","label":"Target reference"},{"id":"source_mel","component":"frontend-logmel","detail":"Whisper content features use source and reference audio."},{"id":"content","component":"encoder-whisper","detail":"Whisper-small content features, not transcript decoding."},{"id":"length","component":"encoder-seed-length"},{"id":"fbank","component":"frontend-kaldi-fbank"},{"id":"speaker","component":"encoder-campplus"},{"id":"mel","component":"frontend-logmel","detail":"Target mel prompt for the acoustic flow."},{"id":"flow","component":"flow-seed-uvit","detail":"Whisper VC checkpoint uses a WaveNet output head after the U-ViT."},{"id":"vocoder","component":"codec-bigvgan"},{"id":"out","kind":"output","label":"22.05 kHz converted voice"}],"edges":[{"from":"source","to":"source_mel"},{"from":"ref","to":"source_mel"},{"from":"source_mel","to":"content"},{"from":"content","to":"length"},{"from":"ref","to":"fbank"},{"from":"fbank","to":"speaker"},{"from":"ref","to":"mel"},{"from":"length","to":"flow"},{"from":"speaker","to":"flow"},{"from":"mel","to":"flow","label":"Reference prompt"},{"from":"flow","to":"vocoder"},{"from":"vocoder","to":"out"}]},{"name":"v1 VC: XLS-R + HiFT","nodes":[{"id":"source","kind":"input","label":"Source recording"},{"id":"ref","kind":"input","label":"Target reference"},{"id":"content","component":"encoder-xls-r","detail":"Source and reference use XLS-R hidden-layer features."},{"id":"length","component":"encoder-seed-length"},{"id":"fbank","component":"frontend-kaldi-fbank"},{"id":"speaker","component":"encoder-campplus"},{"id":"mel","component":"frontend-logmel"},{"id":"flow","component":"flow-seed-uvit","detail":"Compact U-ViT checkpoint with MLP output head."},{"id":"vocoder","component":"codec-hift"},{"id":"out","kind":"output","label":"22.05 kHz converted voice"}],"edges":[{"from":"source","to":"content"},{"from":"ref","to":"content"},{"from":"content","to":"length"},{"from":"ref","to":"fbank"},{"from":"fbank","to":"speaker"},{"from":"ref","to":"mel"},{"from":"length","to":"flow"},{"from":"speaker","to":"flow"},{"from":"mel","to":"flow","label":"Reference prompt"},{"from":"flow","to":"vocoder"},{"from":"vocoder","to":"out"}]},{"name":"v1 SVC: Whisper + pitch + BigVGAN","nodes":[{"id":"source","kind":"input","label":"Source singing"},{"id":"ref","kind":"input","label":"Target reference"},{"id":"content_mel","component":"frontend-logmel"},{"id":"content","component":"encoder-whisper","detail":"Published SVC checkpoint uses Whisper-small content features."},{"id":"pitch","component":"encoder-rmvpe","detail":"Optional F0 conditioning with semitone / target-pitch adjustment."},{"id":"length","component":"encoder-seed-length","detail":"Pitch embedding is combined with content when F0 conditioning is enabled."},{"id":"fbank","component":"frontend-kaldi-fbank"},{"id":"speaker","component":"encoder-campplus"},{"id":"mel","component":"frontend-logmel"},{"id":"flow","component":"flow-seed-uvit"},{"id":"vocoder","component":"codec-bigvgan"},{"id":"out","kind":"output","label":"44.1 kHz converted singing"}],"edges":[{"from":"source","to":"content_mel"},{"from":"ref","to":"content_mel"},{"from":"content_mel","to":"content"},{"from":"content","to":"length"},{"from":"source","to":"pitch","optional":true},{"from":"ref","to":"pitch","optional":true},{"from":"pitch","to":"length","optional":true},{"from":"ref","to":"fbank"},{"from":"fbank","to":"speaker"},{"from":"ref","to":"mel"},{"from":"length","to":"flow"},{"from":"speaker","to":"flow"},{"from":"mel","to":"flow","label":"Reference prompt"},{"from":"flow","to":"vocoder"},{"from":"vocoder","to":"out"}]}],"sources":["https://github.com/Plachtaa/seed-vc"],"packages":[{"id":"seed_vc_mlx_q8_0","display_name":"SeedVC-MLX Q8_0 GGUF","precision":"q8_0"},{"id":"seed_vc_mlx_f16","display_name":"SeedVC-MLX F16 GGUF","precision":"f16"},{"id":"seed_vc_mlx_orig","display_name":"SeedVC-MLX Original-Dtype GGUF","precision":"orig"},{"id":"seed_vc_mlx_safetensors","display_name":"SeedVC-MLX Safetensors","precision":"native"}],"docs":["docs/models/seed_vc.md","docs/audio_tools.md","docs/gguf.md"],"variants":["v2 HuBERT / ASTRAL","v1 Whisper / BigVGAN VC","v1 XLS-R / HiFT VC","v1 Whisper / BigVGAN SVC"],"inputs":[{"type":"audio","label":"Source recording","required":true},{"type":"audio","label":"Target voice reference","required":true}],"detailStatus":"Four implemented routes distinguished; source/reference content, speaker style, mel prompt and singing pitch branches audited.","usageDoc":"docs/models/seed_vc.md"},{"id":"rvc","family":"rvc","name":"RVC","task":"voice-conversion-audio-coding","tasks":["vc"],"summary":"RVC conditions a VITS-derived synthesizer on HuBERT content features and a trained voice identity. Optional retrieval blends source features with indexed training features. F0 checkpoints add RMVPE pitch to the acoustic prior and harmonic excitation to an NSF/HiFi-GAN decoder; non-F0 checkpoints omit those branches. A reference voice recording is not its zero-shot speaker input.","routes":[{"name":"F0-conditioned VC / SVC","nodes":[{"id":"source","kind":"input","label":"Source audio"},{"id":"voice","kind":"input","label":"Trained voice / speaker ID"},{"id":"index","kind":"input","label":"Feature index","optional":true},{"id":"content","component":"encoder-hubert-rvc"},{"id":"retrieval","component":"dsp-retrieval-blend"},{"id":"pitch","component":"encoder-rmvpe"},{"id":"prior","component":"encoder-rvc-prior"},{"id":"flow","component":"flow-rvc-coupling"},{"id":"decoder","component":"codec-rvc-nsf"},{"id":"out","kind":"output","label":"Converted waveform"}],"edges":[{"from":"source","to":"content"},{"from":"content","to":"retrieval"},{"from":"index","to":"retrieval","optional":true},{"from":"source","to":"pitch"},{"from":"retrieval","to":"prior"},{"from":"pitch","to":"prior","label":"Coarse F0"},{"from":"prior","to":"flow"},{"from":"voice","to":"flow"},{"from":"flow","to":"decoder"},{"from":"voice","to":"decoder"},{"from":"pitch","to":"decoder","label":"Continuous F0"},{"from":"decoder","to":"out"}]},{"name":"Non-F0 voice checkpoint","nodes":[{"id":"source","kind":"input","label":"Source audio"},{"id":"voice","kind":"input","label":"Trained voice / speaker ID"},{"id":"index","kind":"input","label":"Feature index","optional":true},{"id":"content","component":"encoder-hubert-rvc"},{"id":"retrieval","component":"dsp-retrieval-blend"},{"id":"prior","component":"encoder-rvc-prior","detail":"Pitch embedding is absent for non-F0 checkpoints."},{"id":"flow","component":"flow-rvc-coupling"},{"id":"decoder","component":"codec-hifigan-latent","detail":"Trained speaker conditioning is added to the generator; no harmonic excitation branch."},{"id":"out","kind":"output","label":"Converted waveform"}],"edges":[{"from":"source","to":"content"},{"from":"content","to":"retrieval"},{"from":"index","to":"retrieval","optional":true},{"from":"retrieval","to":"prior"},{"from":"prior","to":"flow"},{"from":"voice","to":"flow"},{"from":"flow","to":"decoder"},{"from":"voice","to":"decoder"},{"from":"decoder","to":"out"}]}],"sources":["https://github.com/RVC-Project/Retrieval-based-Voice-Conversion-WebUI"],"packages":[{"id":"rvc_f16","display_name":"RVC F16 GGUF","precision":"f16"}],"docs":["docs/audio_tools.md","docs/gguf.md"],"variants":["RVC"],"inputs":[{"type":"audio","label":"Source speech / singing","required":true},{"type":"control","label":"Trained voice checkpoint / speaker ID","required":true},{"type":"data","label":"Retrieval feature index","required":false},{"type":"control","label":"Pitch shift / supplied pitch","required":false}],"detailStatus":"v1/v2 HuBERT selection, optional retrieval, trained speaker identity and F0 versus non-F0 synthesis distinguished.","usageDoc":"docs/audio_tools.md#rvc"},{"id":"tone_color_vc","family":"tone_color_vc","name":"Tone Color VC","task":"voice-conversion-audio-coding","tasks":["vc"],"summary":"Tone Color VC is the converter component of OpenVoice. A posterior encoder analyzes source spectra; a shared CNN/GRU reference encoder extracts both speaker identities. Forward source-conditioned coupling and inverse target-conditioned coupling transfer tone color before waveform decoding. No text-to-speech backbone is part of this model.","routes":[{"name":"Tone-color conversion","nodes":[{"id":"source","kind":"input","label":"Source recording"},{"id":"ref","kind":"input","label":"Target reference"},{"id":"source_spec","component":"frontend-linear-spectrum"},{"id":"target_spec","component":"frontend-linear-spectrum"},{"id":"source_id","component":"encoder-cnn-gru-speaker"},{"id":"target_id","component":"encoder-cnn-gru-speaker"},{"id":"posterior","component":"encoder-wavenet-posterior"},{"id":"forward","component":"flow-tone-coupling","detail":"Forward transform conditioned on source speaker."},{"id":"inverse","component":"flow-tone-coupling","detail":"Inverse transform conditioned on target speaker, using the same coupling weights."},{"id":"decoder","component":"codec-hifigan-latent"},{"id":"out","kind":"output","label":"Converted waveform"}],"edges":[{"from":"source","to":"source_spec"},{"from":"ref","to":"target_spec"},{"from":"source_spec","to":"source_id"},{"from":"target_spec","to":"target_id"},{"from":"source_spec","to":"posterior"},{"from":"posterior","to":"forward"},{"from":"source_id","to":"forward"},{"from":"forward","to":"inverse"},{"from":"target_id","to":"inverse"},{"from":"inverse","to":"decoder"},{"from":"decoder","to":"out"}]}],"sources":["https://github.com/myshell-ai/OpenVoice"],"packages":[{"id":"tone_color_vc_f32","display_name":"Tone Color VC F32 GGUF","precision":"f32"},{"id":"tone_color_vc_f16","display_name":"Tone Color VC F16 GGUF","precision":"f16"}],"docs":["docs/models/tone_color_vc.md"],"variants":["Tone Color VC"],"inputs":[{"type":"audio","label":"Source recording","required":true},{"type":"audio","label":"Target voice reference","required":true}],"detailStatus":"Source and target embeddings, posterior sampling and both flow directions shown separately; no diffusion sampler implied.","usageDoc":"docs/models/tone_color_vc.md"},{"id":"miocodec","family":"miocodec","name":"MioCodec","task":"voice-conversion-audio-coding","tasks":["vc","s2s"],"summary":"MioCodec converts voices by combining source content representations with a reference global embedding. WavLM features feed separate content and global branches: a local Transformer with finite scalar quantization for content, and ConvNeXt with attentive statistics pooling for the voice. A conditioned Transformer and convolutional upsampler predict spectral frames for inverse STFT.","routes":[{"name":"Content / global voice conversion","nodes":[{"id":"source","kind":"input","label":"Source content audio"},{"id":"ref","kind":"input","label":"Voice reference"},{"id":"source_ssl","component":"encoder-wavlm","detail":"Configured content feature layers."},{"id":"ref_ssl","component":"encoder-wavlm","detail":"Configured global feature layer mixture."},{"id":"content","component":"encoder-mio-content"},{"id":"global","component":"encoder-mio-global"},{"id":"decoder","component":"codec-miocodec"},{"id":"out","kind":"output","label":"Converted waveform"}],"edges":[{"from":"source","to":"source_ssl"},{"from":"ref","to":"ref_ssl"},{"from":"source_ssl","to":"content"},{"from":"ref_ssl","to":"global"},{"from":"content","to":"decoder","label":"Quantized content"},{"from":"global","to":"decoder","label":"Global condition"},{"from":"decoder","to":"out"}]}],"sources":["https://huggingface.co/Aratako/MioCodec-25Hz-44.1kHz-v2"],"packages":[{"id":"miocodec_q8_0","display_name":"MioCodec 25Hz 44.1kHz v2 Q8_0 GGUF","precision":"q8_0"},{"id":"miocodec_f16","display_name":"MioCodec 25Hz 44.1kHz v2 F16 GGUF","precision":"f16"},{"id":"miocodec_orig","display_name":"MioCodec 25Hz 44.1kHz v2 Original-Dtype GGUF","precision":"orig"}],"docs":["docs/audio_tools.md","docs/gguf.md"],"variants":["MioCodec 25Hz 44.1kHz v2"],"inputs":[{"type":"audio","label":"Source content","required":true},{"type":"audio","label":"Global voice reference","required":true}],"detailStatus":"The current audio.cpp voice-conversion route is shown; the global encoder is ConvNeXt, not a Transformer.","usageDoc":"docs/audio_tools.md#miocodec"},{"id":"htdemucs","family":"htdemucs","name":"HTDemucs","task":"separation-restoration-enhancement","tasks":["sep"],"summary":"HTDemucs processes the waveform and complex spectrogram in parallel. Its hybrid Transformer alternates within-domain self-attention with cross-attention between domains. Convolutional decoders reconstruct and sum waveform-domain and spectral-domain estimates for each stem.","routes":[{"name":"Four-stem separation","nodes":[{"id":"audio","kind":"input","label":"Music mixture"},{"id":"stft","component":"dsp-complex-stft"},{"id":"waveenc","component":"encoder-demucs-waveform"},{"id":"specenc","component":"encoder-demucs-spectrum"},{"id":"transformer","component":"encoder-hybrid-transformer"},{"id":"wavedec","component":"codec-demucs-waveform"},{"id":"specdec","component":"codec-demucs-spectrum"},{"id":"istft","component":"dsp-inverse-stft"},{"id":"sum","component":"dsp-demucs-sum"},{"id":"out","kind":"output","label":"Vocals / drums / bass / other"}],"edges":[{"from":"audio","to":"stft"},{"from":"audio","to":"waveenc"},{"from":"stft","to":"specenc"},{"from":"waveenc","to":"transformer"},{"from":"specenc","to":"transformer"},{"from":"transformer","to":"wavedec"},{"from":"transformer","to":"specdec"},{"from":"waveenc","to":"wavedec","label":"Skip features"},{"from":"specenc","to":"specdec","label":"Skip features"},{"from":"specdec","to":"istft"},{"from":"istft","to":"sum"},{"from":"wavedec","to":"sum"},{"from":"sum","to":"out"}]}],"sources":["https://github.com/facebookresearch/demucs"],"packages":[{"id":"htdemucs_q8_0","display_name":"HTDemucs Q8_0 GGUF","precision":"q8_0"},{"id":"htdemucs_f16","display_name":"HTDemucs F16 GGUF","precision":"f16"}],"docs":["docs/audio_tools.md","docs/gguf.md"],"variants":["HTDemucs"],"inputs":[{"type":"audio","label":"Music mixture","required":true}],"detailStatus":"Dual-domain encoders, bidirectional cross-domain bottleneck and summed reconstruction described.","usageDoc":"docs/audio_tools.md#htdemucs"},{"id":"htdemucs_6stems","family":"htdemucs_6stems","name":"HTDemucs 6-stem","task":"separation-restoration-enhancement","tasks":["sep"],"summary":"The six-stem HTDemucs checkpoint extends the source outputs to include guitar and piano. It retains the dual waveform/spectrogram encoders, hybrid Transformer and convolutional reconstruction branches.","routes":[{"name":"Six-stem separation","nodes":[{"id":"audio","kind":"input","label":"Music mixture"},{"id":"stft","component":"dsp-complex-stft"},{"id":"waveenc","component":"encoder-demucs-waveform"},{"id":"specenc","component":"encoder-demucs-spectrum"},{"id":"transformer","component":"encoder-hybrid-transformer"},{"id":"wavedec","component":"codec-demucs-waveform"},{"id":"specdec","component":"codec-demucs-spectrum"},{"id":"istft","component":"dsp-inverse-stft"},{"id":"sum","component":"dsp-demucs-sum"},{"id":"out","kind":"output","label":"Vocals / drums / bass / other / guitar / piano"}],"edges":[{"from":"audio","to":"stft"},{"from":"audio","to":"waveenc"},{"from":"stft","to":"specenc"},{"from":"waveenc","to":"transformer"},{"from":"specenc","to":"transformer"},{"from":"transformer","to":"wavedec"},{"from":"transformer","to":"specdec"},{"from":"waveenc","to":"wavedec","label":"Skip features"},{"from":"specenc","to":"specdec","label":"Skip features"},{"from":"specdec","to":"istft"},{"from":"istft","to":"sum"},{"from":"wavedec","to":"sum"},{"from":"sum","to":"out"}]}],"sources":["https://github.com/facebookresearch/demucs"],"packages":[{"id":"htdemucs_6stems_q8_0","display_name":"HTDemucs 6-stem Q8_0 GGUF","precision":"q8_0"},{"id":"htdemucs_6stems_f16","display_name":"HTDemucs 6-stem F16 GGUF","precision":"f16"}],"docs":["docs/audio_tools.md","docs/gguf.md"],"variants":["HTDemucs 6-stem"],"inputs":[{"type":"audio","label":"Music mixture","required":true}],"detailStatus":"Uses the same hybrid topology as four-stem HTDemucs, with a six-source output configuration.","usageDoc":"docs/audio_tools.md#htdemucs-6-stem"},{"id":"bs_roformer","family":"bs_roformer","name":"BS-RoFormer","task":"separation-restoration-enhancement","tasks":["sep"],"summary":"BS-RoFormer projects disjoint frequency bands to a shared feature width, then alternates attention along time and across bands. Band-specific MLPs predict complex masks for source separation.","routes":[{"name":"Music source separation","nodes":[{"id":"audio","kind":"input","label":"Music mixture"},{"id":"stft","component":"dsp-complex-stft"},{"id":"bands","component":"encoder-roformer-band-split","detail":"Disjoint bands with checkpoint-specific widths."},{"id":"encoder","component":"encoder-roformer"},{"id":"head","component":"head-roformer-mask"},{"id":"synthesis","component":"head-complex-mask-synthesis"},{"id":"out","kind":"output","label":"Separated source audio"}],"edges":[{"from":"audio","to":"stft"},{"from":"stft","to":"bands"},{"from":"bands","to":"encoder"},{"from":"encoder","to":"head"},{"from":"head","to":"synthesis"},{"from":"stft","to":"synthesis","label":"Input spectrum"},{"from":"synthesis","to":"out"}]}],"sources":["https://arxiv.org/abs/2309.02612"],"packages":[{"id":"bs_roformer_q8_0","display_name":"BS-RoFormer ep368 Q8_0 GGUF","precision":"q8_0"}],"docs":["docs/audio_tools.md","docs/gguf.md","tests/bs_roformer/README.md"],"variants":["BS-RoFormer ep368"],"inputs":[{"type":"audio","label":"Music mixture","required":true}],"detailStatus":"Band projection, axial rotary attention and complex mask reconstruction described.","usageDoc":"docs/audio_tools.md#bs-roformer"},{"id":"mel_band_roformer","family":"mel_band_roformer","name":"Mel-Band RoFormer","task":"separation-restoration-enhancement","tasks":["sep"],"summary":"Mel-Band RoFormer uses overlapping mel-spaced bands rather than BS-RoFormer's disjoint partition. The axial Transformer and mask-estimation pattern is shared; contributions to overlapping frequency bins are averaged before applying the mask.","routes":[{"name":"Music source separation","nodes":[{"id":"audio","kind":"input","label":"Music mixture"},{"id":"stft","component":"dsp-complex-stft"},{"id":"bands","component":"encoder-roformer-band-split","detail":"Overlapping mel-spaced bands; a frequency bin can enter multiple band embeddings."},{"id":"encoder","component":"encoder-roformer"},{"id":"head","component":"head-roformer-mask","detail":"Average overlapping band contributions per frequency bin before complex masking."},{"id":"synthesis","component":"head-complex-mask-synthesis"},{"id":"out","kind":"output","label":"Separated source audio"}],"edges":[{"from":"audio","to":"stft"},{"from":"stft","to":"bands"},{"from":"bands","to":"encoder"},{"from":"encoder","to":"head"},{"from":"head","to":"synthesis"},{"from":"stft","to":"synthesis","label":"Input spectrum"},{"from":"synthesis","to":"out"}]}],"sources":["https://arxiv.org/abs/2310.01809"],"packages":[{"id":"mel_band_roformer_q8_0","display_name":"Mel-Band RoFormer Q8_0 GGUF","precision":"q8_0"},{"id":"mel_band_roformer_f16","display_name":"Mel-Band RoFormer F16 GGUF","precision":"f16"},{"id":"mel_band_roformer_safetensors","display_name":"Mel RoFormer Safetensors","precision":"native"}],"docs":["docs/audio_tools.md","docs/gguf.md"],"variants":["Mel-Band RoFormer"],"inputs":[{"type":"audio","label":"Music mixture","required":true}],"detailStatus":"Overlapping mel-band projection, axial attention and mask averaging described.","usageDoc":"docs/audio_tools.md#mel-band-roformer"},{"id":"sam_audio","family":"sam_audio","name":"SAM Audio","task":"separation-restoration-enhancement","tasks":["s2s"],"summary":"SAM Audio separates a prompted sound with continuous DAC-VAE latents and a flow-matching DiT. T5 supplies text memory; optional Perception Encoder features and temporal anchor embeddings condition the same generator. Target and residual are both generated and decoded, rather than obtaining the residual by waveform subtraction.","routes":[{"name":"Prompt-conditioned separation","nodes":[{"id":"audio","kind":"input","label":"Mixture audio"},{"id":"text","kind":"input","label":"Sound description"},{"id":"visual","kind":"input","label":"Image / video","optional":true},{"id":"spans","kind":"input","label":"Time spans","optional":true},{"id":"encode","component":"encoder-dac-vae"},{"id":"t5","component":"encoder-sam-t5"},{"id":"vision","component":"encoder-perception-vision"},{"id":"anchors","component":"encoder-temporal-anchors"},{"id":"flow","component":"flow-sam-dit"},{"id":"decode","component":"codec-sam-dac-vae"},{"id":"out","kind":"output","label":"Target + residual audio"}],"edges":[{"from":"audio","to":"encode"},{"from":"text","to":"t5"},{"from":"visual","to":"vision","optional":true},{"from":"spans","to":"anchors","optional":true},{"from":"encode","to":"flow","label":"Mixture latents"},{"from":"t5","to":"flow","label":"Text memory"},{"from":"vision","to":"flow","optional":true},{"from":"anchors","to":"flow","optional":true},{"from":"flow","to":"decode","label":"Two latent streams"},{"from":"decode","to":"out"}]}],"sources":["https://github.com/facebookresearch/sam-audio"],"packages":[{"id":"sam_audio_small_f32","display_name":"SAM Audio Small F32 GGUF","precision":"f32"},{"id":"sam_audio_small_bf16","display_name":"SAM Audio Small BF16 GGUF","precision":"bf16"},{"id":"sam_audio_small_q8_0","display_name":"SAM Audio Small Q8_0 GGUF","precision":"q8_0"},{"id":"sam_audio_base_f32","display_name":"SAM Audio Base F32 GGUF","precision":"f32"},{"id":"sam_audio_base_bf16","display_name":"SAM Audio Base BF16 GGUF","precision":"bf16"},{"id":"sam_audio_base_q8_0","display_name":"SAM Audio Base Q8_0 GGUF","precision":"q8_0"},{"id":"sam_audio_large_f32","display_name":"SAM Audio Large F32 GGUF","precision":"f32"},{"id":"sam_audio_large_bf16","display_name":"SAM Audio Large BF16 GGUF","precision":"bf16"},{"id":"sam_audio_large_q8_0","display_name":"SAM Audio Large Q8_0 GGUF","precision":"q8_0"}],"docs":["docs/models/sam_audio.md","docs/audio_tools.md","docs/gguf.md"],"variants":["SAM Audio Small","SAM Audio Base","SAM Audio Large"],"inputs":[{"type":"audio","label":"Mixture recording","required":true},{"type":"text","label":"Target sound description","required":true},{"type":"image/video","label":"Visual conditioning","required":false},{"type":"control","label":"Positive / negative time spans","required":false}],"detailStatus":"Audio, text, optional visual and temporal conditioning traced separately; nested DiT and codec blocks described. Automatic span prediction and candidate reranking are not shown as implemented routes.","usageDoc":"docs/models/sam_audio.md"},{"id":"apollo","family":"apollo","name":"Apollo","task":"separation-restoration-enhancement","tasks":["s2s"],"summary":"Apollo restores degraded music in the complex STFT domain. Rotary attention exchanges information across frequency bands, while residual temporal convolutions model each band's evolution. The output is a predicted spectrum, not a mask multiplied into the input.","routes":[{"name":"Music restoration","nodes":[{"id":"audio","kind":"input","label":"Degraded music"},{"id":"stft","component":"dsp-complex-stft"},{"id":"bands","component":"encoder-apollo-band-projection"},{"id":"network","component":"encoder-apollo-rope-tcn"},{"id":"head","component":"head-apollo-spectrum"},{"id":"istft","component":"dsp-inverse-stft"},{"id":"out","kind":"output","label":"Restored music"}],"edges":[{"from":"audio","to":"stft"},{"from":"stft","to":"bands"},{"from":"bands","to":"network"},{"from":"network","to":"head"},{"from":"head","to":"istft"},{"from":"istft","to":"out"}]}],"sources":["https://huggingface.co/JusperLee/Apollo"],"packages":[{"id":"apollo_orig","display_name":"Apollo GGUF F32","precision":"f32"}],"docs":[],"variants":["Apollo"],"inputs":[{"type":"audio","label":"Degraded music","required":true}],"detailStatus":"Band normalization, rotary band attention, temporal convolutions and direct spectrum prediction described.","usageDoc":"docs/models/apollo.md"},{"id":"universr","family":"universr","name":"UniverSR","task":"separation-restoration-enhancement","tasks":["s2s"],"summary":"UniverSR generates missing high-frequency complex STFT content using flow matching. ConvNeXt features from the low band and sample-rate embeddings condition a ConvNeXt U-Net. The observed low band is retained during reconstruction; no neural vocoder is required.","routes":[{"name":"Audio super-resolution","nodes":[{"id":"audio","kind":"input","label":"Bandwidth-limited audio"},{"id":"rate","kind":"input","label":"Input sample rate"},{"id":"noise","kind":"input","label":"Gaussian noise"},{"id":"features","component":"dsp-universr-analysis"},{"id":"condition","component":"encoder-universr-convnext"},{"id":"flow","component":"flow-universr-unet"},{"id":"combine","component":"dsp-universr-synthesis"},{"id":"out","kind":"output","label":"48 kHz audio"}],"edges":[{"from":"audio","to":"features"},{"from":"rate","to":"features"},{"from":"features","to":"condition"},{"from":"rate","to":"condition"},{"from":"condition","to":"flow"},{"from":"rate","to":"flow"},{"from":"noise","to":"flow"},{"from":"flow","to":"combine","label":"Generated high band"},{"from":"features","to":"combine","label":"Observed low band"},{"from":"combine","to":"out"}]}],"sources":["https://huggingface.co/woongzip1/universr-audio"],"packages":[{"id":"universr_audio_orig","display_name":"UniverSR Audio GGUF F32","precision":"f32"},{"id":"universr_speech_orig","display_name":"UniverSR Speech GGUF F32","precision":"f32"}],"docs":[],"variants":["UniverSR Audio","UniverSR Speech"],"inputs":[{"type":"audio","label":"Bandwidth-limited audio","required":true},{"type":"control","label":"Input bandwidth / sample rate","required":true}],"detailStatus":"ConvNeXt conditioning, time-conditioned flow U-Net and low/high-band reconstruction described.","usageDoc":"docs/models/universr.md"},{"id":"audiosr","family":"audiosr","name":"AudioSR","task":"separation-restoration-enhancement","tasks":["s2s"],"summary":"AudioSR encodes a low-pass mel spectrogram into continuous latents. A conditional diffusion U-Net generates high-bandwidth latents, a VAE decoder reconstructs mel features, and HiFi-GAN synthesizes audio. Reference low bands are restored before and after vocoding.","routes":[{"name":"Audio super-resolution","nodes":[{"id":"audio","kind":"input","label":"Low-bandwidth audio"},{"id":"frontend","component":"frontend-audiosr-bandlimit"},{"id":"encoder","component":"encoder-audiosr-vae"},{"id":"flow","component":"flow-audiosr-unet"},{"id":"decoder","component":"codec-audiosr-vae"},{"id":"mel_fix","component":"dsp-audiosr-preserve","detail":"Before vocoding: replace low mel bins with the prepared reference."},{"id":"vocoder","component":"codec-hifi-gan"},{"id":"wave_fix","component":"dsp-audiosr-preserve","detail":"After vocoding: match cutoff energy, restore reference low STFT bins, invert STFT, normalize and trim."},{"id":"out","kind":"output","label":"48 kHz waveform"}],"edges":[{"from":"audio","to":"frontend"},{"from":"frontend","to":"encoder"},{"from":"encoder","to":"flow","label":"Concatenated condition"},{"from":"flow","to":"decoder"},{"from":"decoder","to":"mel_fix"},{"from":"frontend","to":"mel_fix","label":"Low mel bins"},{"from":"mel_fix","to":"vocoder"},{"from":"vocoder","to":"wave_fix"},{"from":"frontend","to":"wave_fix","label":"Low-band waveform"},{"from":"wave_fix","to":"out"}]}],"sources":["https://github.com/haoheliu/versatile_audio_super_resolution"],"packages":[{"id":"audiosr_basic_f32","display_name":"AudioSR Basic F32 GGUF","precision":"f32"}],"docs":["docs/models/audiosr.md","docs/gguf.md"],"variants":["AudioSR Basic"],"inputs":[{"type":"audio","label":"Low-bandwidth recording","required":true}],"detailStatus":"Conditioning, mel VAE, diffusion and low-band preservation branches audited. No text or reference-speaker input.","usageDoc":"docs/models/audiosr.md"},{"id":"nemotron_3_diar","family":"nemotron_3_diar","name":"Nemotron 3 Diarization","task":"diarization-activity-detection-alignment","tasks":["diar"],"summary":"Nemotron 3 Diarization stacks mel frames and uses a rotary Transformer, not a FastConformer. Arrival-order speaker memory and recent context are prepended before encoding; an upsampling head predicts independent speaker activity probabilities.","routes":[{"name":"Speaker diarization","nodes":[{"id":"audio","kind":"input","label":"Multi-speaker audio"},{"id":"mel","component":"frontend-logmel"},{"id":"encoder","component":"encoder-rope-transformer"},{"id":"head","component":"head-nemotron-speaker-upsampling"},{"id":"post","component":"dsp-speaker-segmentation"},{"id":"out","kind":"output","label":"Speaker turns / frame probabilities"}],"edges":[{"from":"audio","to":"mel"},{"from":"mel","to":"encoder"},{"from":"encoder","to":"head"},{"from":"head","to":"post"},{"from":"post","to":"out"}]}],"sources":["https://huggingface.co/nvidia/Nemotron-3-Diarization"],"packages":[{"id":"nemotron_3_diar_bf16","display_name":"Nemotron 3 Diarization BF16 GGUF","precision":"bf16"},{"id":"nemotron_3_diar_q8_0","display_name":"Nemotron 3 Diarization Q8_0 GGUF","precision":"q8_0"}],"docs":["docs/models/nemotron_3_diar.md","docs/speech_analysis.md","docs/gguf.md"],"variants":["Nemotron 3 Diarization"],"inputs":[{"type":"audio","label":"Multi-speaker audio","required":true}],"detailStatus":"Stacked features, cache placement and upsampled speaker head audited.","usageDoc":"docs/models/nemotron_3_diar.md"},{"id":"sortformer_diar","family":"sortformer_diar","name":"Sortformer v1","task":"diarization-activity-detection-alignment","tasks":["diar"],"summary":"Sortformer v1 combines a FastConformer acoustic encoder with a second Transformer stack and sigmoid speaker classifiers. Speaker channels follow arrival order; this is not embedding clustering or speech transcription.","routes":[{"name":"Offline speaker diarization","nodes":[{"id":"audio","kind":"input","label":"Multi-speaker audio"},{"id":"mel","component":"frontend-logmel"},{"id":"encoder","component":"encoder-fastconformer"},{"id":"transformer","component":"encoder-sortformer-transformer"},{"id":"head","component":"head-sortformer"},{"id":"post","component":"dsp-speaker-segmentation"},{"id":"out","kind":"output","label":"Speaker turns"}],"edges":[{"from":"audio","to":"mel"},{"from":"mel","to":"encoder"},{"from":"encoder","to":"transformer"},{"from":"transformer","to":"head"},{"from":"head","to":"post"},{"from":"post","to":"out"}]}],"sources":["https://huggingface.co/nvidia/diar_sortformer_4spk-v1"],"packages":[{"id":"sortformer_diar_4spk_v1_q8_0","display_name":"Sortformer Diar 4spk v1 Q8_0 GGUF","precision":"q8_0"},{"id":"sortformer_diar_4spk_v1_f16","display_name":"Sortformer Diar 4spk v1 F16 GGUF","precision":"f16"},{"id":"sortformer_diar_4spk_v1_safetensors","display_name":"Sortformer Diar 4spk v1 Safetensors","precision":"native"}],"docs":["docs/audio_tools.md","docs/gguf.md"],"variants":["Sortformer Diar 4spk v1"],"inputs":[{"type":"audio","label":"Multi-speaker audio","required":true}],"detailStatus":"Offline FastConformer, post-norm Transformer and speaker head audited.","usageDoc":"docs/speech_analysis.md#sortformer-diarization"},{"id":"sortformer_diar_v2","family":"sortformer_diar_v2","name":"Sortformer v2.1","task":"diarization-activity-detection-alignment","tasks":["diar"],"summary":"Streaming Sortformer retains representative speaker frames and recent FIFO context. These are combined with the current subsampled chunk before Conformer and Transformer encoding to keep speaker identities consistent across chunks.","routes":[{"name":"Streaming speaker diarization","nodes":[{"id":"audio","kind":"input","label":"Audio chunks"},{"id":"mel","component":"frontend-logmel"},{"id":"encoder","component":"encoder-sortformer-streaming-conformer"},{"id":"transformer","component":"encoder-sortformer-transformer"},{"id":"head","component":"head-sortformer"},{"id":"post","component":"dsp-speaker-segmentation"},{"id":"out","kind":"output","label":"Incremental speaker turns"}],"edges":[{"from":"audio","to":"mel"},{"from":"mel","to":"encoder"},{"from":"encoder","to":"transformer"},{"from":"transformer","to":"head"},{"from":"head","to":"post"},{"from":"post","to":"out"}]}],"sources":["https://huggingface.co/nvidia/diar_streaming_sortformer_4spk-v2.1"],"packages":[{"id":"sortformer_diar_v2_1_f32_local","display_name":"Sortformer Diar v2.1 F32 local package","precision":"f32"},{"id":"sortformer_diar_v2_1_f32_gguf_local","display_name":"Sortformer Diar v2.1 F32 GGUF local package","precision":"f32"},{"id":"sortformer_diar_v2_1_f16_mixed_gguf_local","display_name":"Sortformer Diar v2.1 mixed F16/F32 GGUF local package","precision":"f16"}],"docs":["docs/community_models/sortformer_diar_v2.md","docs/speech_analysis.md","docs/gguf.md"],"variants":["v2.1"],"inputs":[{"type":"audio","label":"Multi-speaker audio stream","required":true}],"detailStatus":"AOSC context enters before the Conformer stack; speaker-head route audited.","usageDoc":"docs/community_models/sortformer_diar_v2.md"},{"id":"qwen3_forced_aligner","family":"qwen3_forced_aligner","name":"Qwen3 Forced Aligner","task":"diarization-activity-detection-alignment","tasks":["align"],"summary":"Qwen3 Forced Aligner reads audio and an already-known transcript. A Qwen3 causal backbone classifies timestamp placeholders in one prompt pass; it does not autoregressively generate a new transcript.","routes":[{"name":"Word alignment","nodes":[{"id":"audio","kind":"input","label":"Speech audio"},{"id":"text","kind":"input","label":"Transcript + language"},{"id":"mel","component":"frontend-qwen-mel"},{"id":"encoder","component":"encoder-qwen3-asr-encoder"},{"id":"tokens","component":"frontend-qwen-align-prompt"},{"id":"backbone","component":"encoder-qwen3-align-backbone"},{"id":"head","component":"head-timestamp-prediction"},{"id":"out","kind":"output","label":"Word timestamps"}],"edges":[{"from":"audio","to":"mel"},{"from":"mel","to":"encoder"},{"from":"encoder","to":"backbone"},{"from":"text","to":"tokens"},{"from":"tokens","to":"backbone"},{"from":"backbone","to":"head"},{"from":"head","to":"out"}]}],"sources":["https://huggingface.co/Qwen/Qwen3-ForcedAligner-0.6B"],"packages":[{"id":"qwen3_forced_aligner_0_6b_q8_0","display_name":"Qwen3 Forced Aligner 0.6B Q8_0 GGUF","precision":"q8_0"},{"id":"qwen3_forced_aligner_0_6b_f16","display_name":"Qwen3 Forced Aligner 0.6B F16 GGUF","precision":"f16"},{"id":"qwen3_forced_aligner_0_6b_safetensors","display_name":"Qwen3 Forced Aligner 0.6B Safetensors","precision":"native"}],"docs":["docs/models/qwen3.md","docs/gguf.md"],"variants":["Qwen3 Forced Aligner 0.6B"],"inputs":[{"type":"audio","label":"Speech","required":true},{"type":"text","label":"Known transcript","required":true},{"type":"control","label":"Transcript language","required":true}],"detailStatus":"Supplied transcript and timestamp-placeholder classification audited.","usageDoc":"docs/models/qwen3.md#qwen3-forced-aligner"},{"id":"mms_forced_aligner","family":"mms_forced_aligner","name":"MMS Forced Aligner","task":"diarization-activity-detection-alignment","tasks":["align"],"summary":"MMS encodes the waveform with Wav2Vec2 and predicts framewise character scores. The supplied transcript constrains a CTC alignment path; it is not used as an audio-encoder prompt.","routes":[{"name":"CTC forced alignment","nodes":[{"id":"audio","kind":"input","label":"Speech audio"},{"id":"text","kind":"input","label":"Known transcript"},{"id":"encoder","component":"encoder-wav2vec2"},{"id":"head","component":"head-ctc-emissions"},{"id":"norm","component":"frontend-mms-transcript"},{"id":"align","component":"dsp-ctc-alignment"},{"id":"out","kind":"output","label":"Word timestamps"}],"edges":[{"from":"audio","to":"encoder"},{"from":"encoder","to":"head"},{"from":"head","to":"align","label":"Frame log-probabilities"},{"from":"text","to":"norm"},{"from":"norm","to":"align","label":"Target label sequence"},{"from":"align","to":"out"}]}],"sources":["https://huggingface.co/MahmoudAshraf/mms-300m-1130-forced-aligner/blob/main/config.json"],"packages":[{"id":"mms_forced_aligner_300m_f16","display_name":"Meta MMS-300M Forced Aligner F16 GGUF","precision":"f16"},{"id":"mms_forced_aligner_300m_safetensors","display_name":"Meta MMS-300M Forced Aligner Safetensors","precision":"native"},{"id":"mms_forced_aligner_300m_q8_0","display_name":"Meta MMS-300M Forced Aligner Q8_0 GGUF","precision":"q8_0"}],"docs":["docs/community_models/mms_forced_aligner.md","docs/speech_analysis.md","docs/gguf.md"],"variants":["Meta MMS-300M Forced Aligner"],"inputs":[{"type":"audio","label":"Speech","required":true},{"type":"text","label":"Known transcript","required":true}],"detailStatus":"Waveform encoder and transcript-constrained CTC alignment audited.","usageDoc":"docs/community_models/mms_forced_aligner.md"},{"id":"silero_vad","family":"silero_vad","name":"Silero VAD","task":"diarization-activity-detection-alignment","tasks":["vad"],"summary":"Silero VAD computes magnitude spectral features, compresses them with convolutions and carries an LSTM state between windows. A sigmoid head yields speech probability, which is thresholded into speech intervals.","routes":[{"name":"Voice activity detection","nodes":[{"id":"audio","kind":"input","label":"Audio windows"},{"id":"frontend","component":"dsp-stft-magnitude"},{"id":"encoder","component":"encoder-silero-cnn-lstm"},{"id":"head","component":"head-silero-probability"},{"id":"post","component":"dsp-vad-segmentation"},{"id":"out","kind":"output","label":"Speech intervals"}],"edges":[{"from":"audio","to":"frontend"},{"from":"frontend","to":"encoder"},{"from":"encoder","to":"head"},{"from":"head","to":"post"},{"from":"post","to":"out"}]}],"sources":["https://github.com/snakers4/silero-vad"],"packages":[],"docs":[],"variants":["Bundled Silero VAD"],"inputs":[{"type":"audio","label":"Audio","required":true}],"detailStatus":"Spectral CNN, recurrent state and speech probability route audited.","usageDoc":"docs/speech_analysis.md#silero-vad"},{"id":"marblenet_vad","family":"marblenet_vad","name":"MarbleNet VAD","task":"diarization-activity-detection-alignment","tasks":["vad"],"summary":"MarbleNet uses residual time-channel separable convolutions over log-mel features. A speech/non-speech classifier and temporal postprocessing produce activity intervals, without an AR decoder.","routes":[{"name":"Voice activity detection","nodes":[{"id":"audio","kind":"input","label":"Audio"},{"id":"mel","component":"frontend-logmel"},{"id":"encoder","component":"encoder-marblenet"},{"id":"head","component":"head-binary-speech-classifier"},{"id":"post","component":"dsp-vad-segmentation"},{"id":"out","kind":"output","label":"Speech intervals"}],"edges":[{"from":"audio","to":"mel"},{"from":"mel","to":"encoder"},{"from":"encoder","to":"head"},{"from":"head","to":"post"},{"from":"post","to":"out"}]}],"sources":["https://research.nvidia.com/publication/2020-10_marblenet-deep-1d-time-channel-separable-convolutional-neural-network-voice"],"packages":[],"docs":[],"variants":["Bundled MarbleNet VAD"],"inputs":[{"type":"audio","label":"Audio","required":true}],"detailStatus":"Log-mel, separable convolution and frame classifier route audited.","usageDoc":"docs/speech_analysis.md#marblenet-vad"},{"id":"pulsevad","family":"pulsevad","name":"PulseVAD","task":"diarization-activity-detection-alignment","tasks":["vad"],"summary":"PulseVAD processes short normalized log-mel windows with a tiny depthwise-separable CNN. Temporal mean pooling and a two-class head decide whether a window contains speech.","routes":[{"name":"Windowed voice activity detection","nodes":[{"id":"audio","kind":"input","label":"Audio windows"},{"id":"mel","component":"frontend-logmel","detail":"Pre-emphasis, waveform normalization and per-mel-bin normalization within each window."},{"id":"encoder","component":"encoder-depthwise-separable-cnn"},{"id":"head","component":"head-binary-speech-classifier","detail":"Global mean pooling precedes a two-class linear projection."},{"id":"post","component":"dsp-vad-segmentation"},{"id":"out","kind":"output","label":"Speech intervals"}],"edges":[{"from":"audio","to":"mel"},{"from":"mel","to":"encoder"},{"from":"encoder","to":"head"},{"from":"head","to":"post"},{"from":"post","to":"out"}]}],"sources":["https://github.com/AydinAdnan/PulseVAD"],"packages":[{"id":"pulsevad_2_1k_f32","display_name":"PulseVAD 2.1K GGUF F32","precision":"f32"},{"id":"pulsevad_81k_f32","display_name":"PulseVAD 81K GGUF F32","precision":"f32"}],"docs":["docs/models/pulsevad.md"],"variants":["PulseVAD 2.1K","PulseVAD 81K"],"inputs":[{"type":"audio","label":"Audio","required":true}],"detailStatus":"Normalized mel windows, separable CNN and pooled binary head audited.","usageDoc":"docs/models/pulsevad.md"},{"id":"builtin_audio_utils/rnnoise","family":"builtin_audio_utils","name":"RNNoise","task":"built-in-neural-audio-utilities","tasks":["s2s"],"summary":"RNNoise combines traditional spectral and pitch analysis with a small recurrent network. Its predicted band gains control signal processing; it does not generate discrete audio tokens.","routes":[{"name":"Noise suppression","nodes":[{"id":"audio","kind":"input","label":"Noisy speech"},{"id":"features","component":"dsp-rnnoise-analysis"},{"id":"network","component":"encoder-rnnoise-conv-gru"},{"id":"synthesis","component":"dsp-rnnoise-synthesis"},{"id":"out","kind":"output","label":"Enhanced speech"}],"edges":[{"from":"audio","to":"features"},{"from":"features","to":"network"},{"from":"features","to":"synthesis","label":"Spectrum / pitch"},{"from":"network","to":"synthesis","label":"Band gains"},{"from":"synthesis","to":"out"}]}],"sources":["https://github.com/xiph/rnnoise"],"packages":[],"docs":["docs/audio_tools.md"],"variants":["RNNoise"],"inputs":[{"type":"audio","label":"Noisy speech","required":true}],"detailStatus":"Band features, convolution/GRU predictor and pitch-aware synthesis described.","usageDoc":"docs/audio_tools.md#built-in-audio-utilities"},{"id":"builtin_audio_utils/deepfilternet2","family":"builtin_audio_utils","name":"DeepFilterNet2","task":"built-in-neural-audio-utilities","tasks":["s2s"],"summary":"DeepFilterNet2 uses separate ERB-energy and complex-spectrum feature branches. A convolutional/recurrent network predicts band gains and a short complex filter across frames, followed by inverse spectral synthesis.","routes":[{"name":"Speech enhancement","nodes":[{"id":"audio","kind":"input","label":"Noisy speech"},{"id":"features","component":"dsp-deepfilter-analysis"},{"id":"network","component":"encoder-deepfilter-conv-gru"},{"id":"filter","component":"head-deepfilter-synthesis"},{"id":"out","kind":"output","label":"Enhanced speech"}],"edges":[{"from":"audio","to":"features"},{"from":"features","to":"network"},{"from":"features","to":"filter","label":"Complex spectrum"},{"from":"network","to":"filter","label":"Gains + filter coefficients"},{"from":"filter","to":"out"}]}],"sources":["https://github.com/Rikorose/DeepFilterNet"],"packages":[],"docs":["docs/audio_tools.md"],"variants":["DeepFilterNet2"],"inputs":[{"type":"audio","label":"Noisy speech","required":true}],"detailStatus":"ERB and complex-feature branches, grouped GRU and reconstruction described.","usageDoc":"docs/audio_tools.md#built-in-audio-utilities"},{"id":"builtin_audio_utils/zipenhancer","family":"builtin_audio_utils","name":"ZipEnhancer","task":"built-in-neural-audio-utilities","tasks":["s2s"],"summary":"ZipEnhancer processes a compressed complex spectrogram with a dense convolutional encoder and dual-path Zipformer. Separate magnitude and phase heads reconstruct the enhanced spectrum.","routes":[{"name":"Speech enhancement","nodes":[{"id":"audio","kind":"input","label":"Noisy speech"},{"id":"features","component":"dsp-zipenhancer-analysis"},{"id":"network","component":"encoder-dual-path-zipformer"},{"id":"synthesis","component":"head-magnitude-phase-synthesis"},{"id":"out","kind":"output","label":"Enhanced speech"}],"edges":[{"from":"audio","to":"features"},{"from":"features","to":"network"},{"from":"network","to":"synthesis"},{"from":"synthesis","to":"out"}]}],"sources":["https://zipenhancer.github.io/ZipEnhancer/"],"packages":[],"docs":["docs/audio_tools.md"],"variants":["ZipEnhancer"],"inputs":[{"type":"audio","label":"Noisy speech","required":true}],"detailStatus":"Spectral compression, dual-axis Zipformer and magnitude/phase heads described.","usageDoc":"docs/audio_tools.md#built-in-audio-utilities"},{"id":"builtin_audio_utils/gtcrn","family":"builtin_audio_utils","name":"GTCRN","task":"built-in-neural-audio-utilities","tasks":["s2s"],"summary":"GTCRN combines grouped temporal convolutions with recurrent processing along frequency and time. Its complex mask modifies the input spectrum before inverse STFT, rather than synthesizing speech from tokens.","routes":[{"name":"Speech enhancement","nodes":[{"id":"audio","kind":"input","label":"Noisy speech"},{"id":"stft","component":"dsp-complex-stft"},{"id":"network","component":"encoder-gtcrn"},{"id":"synthesis","component":"head-complex-mask-synthesis"},{"id":"out","kind":"output","label":"Enhanced speech"}],"edges":[{"from":"audio","to":"stft"},{"from":"stft","to":"network"},{"from":"stft","to":"synthesis","label":"Input spectrum"},{"from":"network","to":"synthesis","label":"Complex mask"},{"from":"synthesis","to":"out"}]}],"sources":["https://github.com/Xiaobin-Rong/gtcrn"],"packages":[],"docs":["docs/audio_tools.md"],"variants":["Streaming","DNS3","VCTK"],"inputs":[{"type":"audio","label":"Noisy speech","required":true}],"detailStatus":"Spectral feature stack, grouped temporal blocks and dual-path GRU described.","usageDoc":"docs/audio_tools.md#gtcrn"},{"id":"builtin_audio_utils/flashsr","family":"builtin_audio_utils","name":"FlashSR","task":"built-in-neural-audio-utilities","tasks":["s2s"],"summary":"This FlashSR is the tiny HierSpeech++-derived waveform upsampler by Yatharth Sharma. It converts 16 kHz audio to 48 kHz using residual convolutions and periodic activations, not a diffusion model.","routes":[{"name":"Audio super-resolution","nodes":[{"id":"audio","kind":"input","label":"16 kHz waveform"},{"id":"network","component":"encoder-flashsr-residual-cnn"},{"id":"out","kind":"output","label":"48 kHz waveform"}],"edges":[{"from":"audio","to":"network"},{"from":"network","to":"out"}]}],"sources":["https://github.com/ysharma3501/FlashSR"],"packages":[],"docs":["docs/audio_tools.md"],"variants":["FlashSR"],"inputs":[{"type":"audio","label":"16 kHz audio","required":true}],"detailStatus":"Direct waveform upsampling and parallel residual convolution branches described.","usageDoc":"docs/audio_tools.md#built-in-audio-utilities"}],"diagrams":{"umt5":{"title":"UMT5 scene conditioning","scope":"The encoder uses per-layer relative attention biases and gated GELU feed-forward layers. There is no autoregressive text-decoding loop.","blocks":[{"id":"tok","label":"SentencePiece tokenization","expand":"sentencepiece"},{"id":"embed","label":"Token embeddings"},{"id":"block","label":"Relative-attention encoder blocks","expand":"umt5-block"},{"id":"out","label":"Final RMSNorm"}],"edges":[["tok","embed"],["embed","block"],["block","out"]]},"umt5-block":{"title":"UMT5 encoder block","scope":"Bidirectional attention with bucketed relative-position bias and pre-RMS normalization; the gated feed-forward activation is GELU.","blocks":[{"id":"norm","label":"RMSNorm"},{"id":"attn","label":"Bidirectional relative attention"},{"id":"r1","label":"Residual add"},{"id":"ff","label":"RMSNorm / gated GELU feed-forward"},{"id":"r2","label":"Residual add"}],"edges":[["norm","attn"],["attn","r1"],["r1","ff"],["ff","r2"]],"residuals":[["norm","r1"],["ff","r2"]]},"wan-audio-condition":{"title":"Wan speech-to-motion conditioning","scope":"XLS-R features are aligned to video timing before learned layer mixing. Local tokens condition cross-attention, while a separate global branch produces modulation.","blocks":[{"id":"mix","label":"Weighted XLS-R layer mixture"},{"id":"local","label":"Local causal Conv1d / LayerNorm / SiLU"},{"id":"global","label":"Global causal Conv1d / LayerNorm / SiLU"},{"id":"tokens","label":"Frame-aligned local audio tokens"},{"id":"mod","label":"Global modulation features"}],"edges":[["mix","local"],["mix","global"],["local","tokens"],["global","mod"]]},"wan-s2v":{"title":"LiveAvatar blockwise video flow","scope":"The current block is denoised with reference and history context. Generated blocks become history for later blocks. Audio injection occurs at selected layers, not every layer.","blocks":[{"id":"input","label":"Noisy video / reference / history latent patches"},{"id":"time","label":"Timestep MLP / modulation"},{"id":"text","label":"UMT5 context projection"},{"id":"dit","label":"Wan DiT blocks","expand":"wan-dit-block"},{"id":"audio","label":"Audio cross-attention / global modulation"},{"id":"head","label":"Adaptive norm / velocity / unpatchify"},{"id":"solve","label":"Flow integration / retain history"}],"edges":[["input","dit"],["time","dit"],["text","dit"],["dit","audio"],["audio","head"],["head","solve"]]},"wan-dit-block":{"title":"Wan S2V DiT block","scope":"Time-adaptive LayerNorm surrounds self-attention and the GELU feed-forward residual. Text is supplied by separate cross-attention; audio is injected between selected blocks.","blocks":[{"id":"norm","label":"Time-adaptive LayerNorm"},{"id":"attn","label":"3D RoPE / Q/K-normalized self-attention"},{"id":"r1","label":"Time gate + residual"},{"id":"cross","label":"LayerNorm / text cross-attention"},{"id":"r2","label":"Residual add"},{"id":"ff","label":"Time-adaptive LayerNorm / GELU FFN"},{"id":"r3","label":"Time gate + residual"}],"edges":[["norm","attn"],["attn","r1"],["r1","cross"],["cross","r2"],["r2","ff"],["ff","r3"]],"residuals":[["norm","r1"],["cross","r2"],["ff","r3"]]},"wan-vae-encoder":{"title":"Wan video compression","scope":"Temporal convolutions are causal; spatial convolutions operate on each frame grid. Reference images use the same latent representation as video history.","blocks":[{"id":"conv","label":"Causal 3D input convolution"},{"id":"down","label":"Residual stages / spatial-temporal downsampling","expand":"wan-vae-residual"},{"id":"mid","label":"Residual / spatial attention / residual"},{"id":"out","label":"Latent distribution projection"}],"edges":[["conv","down"],["down","mid"],["mid","out"]]},"wan-vae-decoder":{"title":"Wan video reconstruction","scope":"Causal convolution caches retain preceding frame context across decode chunks. Spatial attention is distinct from the denoiser's 3D rotary attention.","blocks":[{"id":"conv","label":"Latent input convolution"},{"id":"mid","label":"Residual / spatial attention / residual"},{"id":"up","label":"Residual stages / spatial-temporal upsampling","expand":"wan-vae-residual"},{"id":"out","label":"RMSNorm / SiLU / RGB convolution"}],"edges":[["conv","mid"],["mid","up"],["up","out"]]},"wan-vae-residual":{"title":"Wan causal 3D residual block","scope":"Channel normalization and SiLU precede each causal 3D convolution. The shortcut is projected when channel dimensions differ.","blocks":[{"id":"n1","label":"RMSNorm / SiLU"},{"id":"c1","label":"Causal 3D convolution"},{"id":"n2","label":"RMSNorm / SiLU"},{"id":"c2","label":"Causal 3D convolution"},{"id":"out","label":"Residual add / projected shortcut"}],"edges":[["n1","c1"],["c1","n2"],["n2","c2"],["c2","out"]],"residuals":[["n1","out"]]},"h3-text":{"title":"H3 prompt encoding","scope":"Only text is passed to the Qwen3-VL backbone in this port. Causal hidden states condition the DiT; no text-generation loop or vision tower is needed.","blocks":[{"id":"embed","label":"Prompt token embeddings"},{"id":"qwen","label":"Qwen3-VL causal text blocks","expand":"qwen3-layer"},{"id":"features","label":"Prompt hidden features"}],"edges":[["embed","qwen"],["qwen","features"]]},"h3-dit":{"title":"H3 joint audiovisual flow","scope":"Text features and both latent modalities are refined and combined for joint attention. Audio and video use separate time schedules. Disabling video output skips its decoder, not video latent denoising.","blocks":[{"id":"text","label":"Text feature projection / token refinement"},{"id":"audio","label":"Audio latent projection / token refinement"},{"id":"video","label":"Video latent patches / token refinement"},{"id":"joint","label":"Pack modality tokens + positions"},{"id":"time","label":"Audio / video timestep modulation"},{"id":"dit","label":"Joint rotary DiT stack","expand":"h3-dit-block"},{"id":"head","label":"Modality-specific velocity heads"},{"id":"solve","label":"Coupled audio / video flow solve"}],"edges":[["text","joint"],["audio","joint"],["video","joint"],["joint","dit"],["time","dit"],["dit","head"],["head","solve"]]},"h3-dit-block":{"title":"H3 multimodal DiT block","scope":"Adaptive parameters are selected per modality. Joint Q/K-normalized attention uses multidimensional rotary positions; the feed-forward branch is SwiGLU.","blocks":[{"id":"norm","label":"RMSNorm / modality-specific scale + shift"},{"id":"attn","label":"Q/K-normalized rotary joint attention"},{"id":"r1","label":"Modality gate + residual"},{"id":"ff","label":"Adaptive RMSNorm / SwiGLU"},{"id":"r2","label":"Modality gate + residual"}],"edges":[["norm","attn"],["attn","r1"],["r1","ff"],["ff","r2"]],"residuals":[["norm","r1"],["ff","r2"]]},"h3-audio":{"title":"H3 audio VAE reconstruction","scope":"Stereo latent channels are denormalized and decoded separately through shared decoder weights.","blocks":[{"id":"norm","label":"Latent mean / standard-deviation restoration"},{"id":"conv","label":"Pointwise input projection"},{"id":"decoder","label":"BigVGAN waveform synthesis","expand":"bigvgan"},{"id":"out","label":"Assemble audio channels"}],"edges":[["norm","conv"],["conv","decoder"],["decoder","out"]]},"h3-video":{"title":"H3 Transformer video VAE","scope":"Latent patches and learned register tokens are decoded by a rotary Transformer. Register outputs are discarded before spatial-temporal RGB unpatchification.","blocks":[{"id":"embed","label":"Latent projection + register tokens"},{"id":"stack","label":"Rotary Transformer stack","expand":"h3-video-block"},{"id":"norm","label":"LayerNorm + RGB patch projection"},{"id":"out","label":"Discard registers / unpatchify frames"}],"edges":[["embed","stack"],["stack","norm"],["norm","out"]]},"h3-video-block":{"title":"H3 video VAE Transformer block","scope":"Q/K normalization, rotary attention, SwiGLU and learned residual scales reconstruct video features without a flow timestep.","blocks":[{"id":"norm","label":"RMSNorm"},{"id":"attn","label":"Q/K-normalized rotary attention"},{"id":"r1","label":"Learned scale + residual"},{"id":"ff","label":"RMSNorm / SwiGLU"},{"id":"r2","label":"Learned scale + residual"}],"edges":[["norm","attn"],["attn","r1"],["r1","ff"],["ff","r2"]],"residuals":[["norm","r1"],["ff","r2"]]},"ace-text":{"title":"ACE-Step Qwen3 text encoder","scope":"The full prompt uses hidden states; lyrics start from the same tokenizer's embeddings and enter a separate lyric encoder. This checkpoint is independent of the optional AR planner.","blocks":[{"id":"embed","label":"Token embeddings"},{"id":"stack","label":"Qwen3 Transformer blocks","expand":"qwen3-layer"},{"id":"norm","label":"Final RMSNorm / prompt features"},{"id":"lyrics","label":"Lyric token embeddings"}],"edges":[["embed","stack"],["stack","norm"],["embed","lyrics"]]},"ace-condition":{"title":"ACE-Step conditioning branches","scope":"Text projection, lyric encoding and timbre encoding create conditioning tokens. XL uses an independent encoder width, CLS timbre pooling and projection into the larger denoiser width.","blocks":[{"id":"text","label":"Text hidden-state projection"},{"id":"lyrics","label":"Lyric embeddings / rotary Transformer","expand":"ace-encoder-block"},{"id":"timbre","label":"Reference latents / rotary Transformer","expand":"ace-encoder-block"},{"id":"pool","label":"Timbre pooling; CLS in XL"},{"id":"pack","label":"Pack text / lyric / timbre context"},{"id":"bridge","label":"XL width bridge when configured"}],"edges":[["text","pack"],["lyrics","pack"],["timbre","pool"],["pool","pack"],["pack","bridge"]]},"ace-encoder-block":{"title":"ACE-Step rotary conditioning block","scope":"Noncausal attention over the configured full or sliding window, with RMS normalization and SwiGLU. This is conditioning, not autoregressive generation.","blocks":[{"id":"norm","label":"RMSNorm"},{"id":"attn","label":"Rotary grouped-query attention"},{"id":"r1","label":"Projection + residual"},{"id":"ff","label":"RMSNorm / SwiGLU"},{"id":"r2","label":"Residual add"}],"edges":[["norm","attn"],["attn","r1"],["r1","ff"],["ff","r2"]],"residuals":[["norm","r1"],["ff","r2"]]},"ace-cover":{"title":"ACE-Step cover tokenization","scope":"Input is continuous VAE latent windows. A learned summary token pools each window before finite scalar quantization.","blocks":[{"id":"proj","label":"Latent projection + summary token"},{"id":"enc","label":"Rotary attention-pooling stack","expand":"ace-encoder-block"},{"id":"select","label":"RMSNorm / select summary"},{"id":"fsq","label":"Low-dimensional projection / FSQ packing"}],"edges":[["proj","enc"],["enc","select"],["select","fsq"]]},"ace-detokenizer":{"title":"ACE-Step latent hints","scope":"Music codes from either the AR planner or cover tokenizer share this decoder. It expands each code into a short latent window, not waveform samples.","blocks":[{"id":"fsq","label":"Unpack FSQ coordinates + projection"},{"id":"repeat","label":"Repeat window / add learned position tokens"},{"id":"enc","label":"Rotary Transformer stack","expand":"ace-encoder-block"},{"id":"out","label":"RMSNorm + latent projection"}],"edges":[["fsq","repeat"],["repeat","enc"],["enc","out"]]},"ace-dit":{"title":"ACE-Step acoustic flow","scope":"Noisy latents and source/hint context are patchified. Text/lyric/timbre context is accessed by cross-attention. Time and solver-step embeddings modulate the denoiser.","blocks":[{"id":"patch","label":"Latent + context concatenation / Conv1d patches"},{"id":"time","label":"Time / step Fourier embeddings + MLP"},{"id":"context","label":"Conditioning tokens"},{"id":"dit","label":"Rotary cross-attention DiT stack","expand":"ace-dit-block"},{"id":"out","label":"Adaptive RMSNorm / transposed-conv unpatchify"},{"id":"solve","label":"Flow integration"}],"edges":[["patch","dit"],["time","dit"],["context","dit"],["dit","out"],["out","solve"]]},"ace-dit-block":{"title":"ACE-Step DiT block","scope":"Self-attention follows the checkpoint's full/sliding schedule. Cross-attention uses normalized Q/K without rotary positions; self-attention uses RoPE.","blocks":[{"id":"norm","label":"Time-adaptive RMSNorm"},{"id":"attn","label":"Q/K-normalized rotary self-attention"},{"id":"r1","label":"Time gate + residual"},{"id":"cross","label":"RMSNorm / conditioning cross-attention"},{"id":"r2","label":"Residual add"},{"id":"ff","label":"Time-adaptive RMSNorm + SwiGLU"},{"id":"r3","label":"Time gate + residual"}],"edges":[["norm","attn"],["attn","r1"],["r1","cross"],["cross","r2"],["r2","ff"],["ff","r3"]],"residuals":[["norm","r1"],["cross","r2"],["ff","r3"]]},"ace-vae-encoder":{"title":"ACE-Step audio compression","scope":"Convolutional VAE encoding supplies continuous source and timbre latents. There is no speech-token codebook in this stage.","blocks":[{"id":"input","label":"Waveform input convolution"},{"id":"res","label":"Periodic-activation residual units"},{"id":"down","label":"Strided downsampling stages"},{"id":"out","label":"Continuous latent projection"}],"edges":[["input","res"],["res","down"],["down","out"]]},"ace-vae-decoder":{"title":"ACE-Step waveform reconstruction","scope":"Each upsampling stage has Snake periodic activation and residual convolutions at multiple dilations.","blocks":[{"id":"input","label":"Latent input convolution"},{"id":"up","label":"Snake / transposed-convolution upsampling"},{"id":"res","label":"Dilated Snake residual units"},{"id":"out","label":"Snake / stereo output convolution"}],"edges":[["input","up"],["up","res"],["res","out"]]},"yue2-ar":{"title":"YuE2 AR branch","scope":"The same branch generates score tokens and semantic tokens in successive stages. Its layer-wise K/V also conditions NAR synthesis. Similar operations to Qwen3 do not imply the same checkpoint.","blocks":[{"id":"embed","label":"Text / ABC / semantic token embeddings"},{"id":"stack","label":"Causal rotary AR blocks","expand":"yue2-ar-block"},{"id":"norm","label":"Final RMSNorm"},{"id":"head","label":"Vocabulary projection + sampling"},{"id":"kv","label":"Layer-wise prefix K/V for NAR"}],"edges":[["embed","stack"],["stack","norm"],["norm","head"],["stack","kv"]]},"yue2-ar-block":{"title":"YuE2 causal AR block","scope":"Causal grouped-query attention with Q/K RMS normalization and rotary positions. The AR branch has its own projections and feed-forward weights.","blocks":[{"id":"norm","label":"RMSNorm + Q/K/V projection"},{"id":"rope","label":"Q/K RMSNorm + RoPE"},{"id":"attn","label":"Causal grouped-query attention"},{"id":"r1","label":"Output projection + residual"},{"id":"ff","label":"RMSNorm + SwiGLU"},{"id":"r2","label":"Residual add"}],"edges":[["norm","rope"],["rope","attn"],["attn","r1"],["r1","ff"],["ff","r2"]],"residuals":[["norm","r1"],["ff","r2"]]},"yue2-nar":{"title":"YuE2 acoustic flow","scope":"Noise latents, time and learned audio positions enter the NAR branch. The flow solver repeatedly predicts velocity; AR prefix states stay fixed during the solve.","blocks":[{"id":"latent","label":"Noise latent projection"},{"id":"time","label":"Time MLP + audio position embeddings"},{"id":"add","label":"Add conditioning"},{"id":"kv","label":"AR prefix keys / values"},{"id":"stack","label":"NAR mixed-attention stack","expand":"yue2-nar-block"},{"id":"out","label":"RMSNorm + latent velocity projection"},{"id":"solve","label":"Flow integration"}],"edges":[["latent","add"],["time","add"],["add","stack"],["kv","stack"],["stack","out"],["out","solve"]]},"yue2-nar-block":{"title":"YuE2 NAR mixed-attention block","scope":"Acoustic queries see the full acoustic sequence and the AR prefix. Prefix K/V come from the AR branch, while acoustic K/V and queries use NAR parameters.","blocks":[{"id":"norm","label":"NAR RMSNorm + Q/K/V projections"},{"id":"rope","label":"Q/K RMSNorm + RoPE"},{"id":"prefix","label":"Cached AR prefix K/V"},{"id":"attn","label":"Attention over prefix + acoustic K/V"},{"id":"r1","label":"NAR output projection + residual"},{"id":"ff","label":"NAR RMSNorm + SwiGLU"},{"id":"r2","label":"Residual add"}],"edges":[["norm","rope"],["rope","attn"],["prefix","attn"],["attn","r1"],["r1","ff"],["ff","r2"]],"residuals":[["norm","r1"],["ff","r2"]]},"firered-understanding":{"title":"FireRedAudio understanding encoder","scope":"This route supplies continuous understanding embeddings, not speech-generation latents. It does not run RedAE.","blocks":[{"id":"mel","label":"Whisper-style log-mel frontend","expand":"logmel"},{"id":"conv","label":"Conv1d / GELU subsampling"},{"id":"pos","label":"Add positional embeddings"},{"id":"enc","label":"Noncausal Transformer stack","expand":"audio-transformer"},{"id":"adapt","label":"Two strided Conv1d adapters"},{"id":"proj","label":"LayerNorm / GELU MLP projection"}],"edges":[["mel","conv"],["conv","pos"],["pos","enc"],["enc","adapt"],["adapt","proj"]]},"qwen35":{"title":"Qwen3.5 hybrid decoder","scope":"The configured layer schedule alternates recurrent linear-attention blocks with full-attention blocks. The two branches shown are alternative layer types, not parallel experts.","blocks":[{"id":"input","label":"Token / audio prompt embeddings"},{"id":"delta","label":"Gated DeltaNet layer","expand":"qwen35-delta"},{"id":"attn","label":"Full-attention layer","expand":"qwen35-attention"},{"id":"norm","label":"Final RMSNorm"},{"id":"out","label":"Text logits / audio conditioning"}],"edges":[["input","delta"],["input","attn"],["delta","norm"],["attn","norm"],["norm","out"]]},"qwen35-delta":{"title":"Qwen3.5 recurrent layer","scope":"Normalized Q/K and learned decay/update gates drive the Gated DeltaNet recurrent state. A separate SiLU output gate follows RMS normalization.","blocks":[{"id":"norm","label":"RMSNorm + projections"},{"id":"conv","label":"Causal depthwise convolution + SiLU"},{"id":"qk","label":"Q/K normalization"},{"id":"gates","label":"Learned decay + update gates"},{"id":"scan","label":"Gated DeltaNet recurrence"},{"id":"out","label":"RMSNorm / SiLU gate / output projection"},{"id":"r1","label":"Residual add"},{"id":"ff","label":"RMSNorm + SwiGLU"},{"id":"r2","label":"Residual add"}],"edges":[["norm","conv"],["conv","qk"],["qk","scan"],["norm","gates"],["gates","scan"],["scan","out"],["out","r1"],["r1","ff"],["ff","r2"]],"residuals":[["norm","r1"],["ff","r2"]]},"qwen35-attention":{"title":"Qwen3.5 full-attention layer","scope":"Q/K normalization and partial rotary positions precede causal grouped-query attention. A sigmoid gate modulates attention output before projection.","blocks":[{"id":"norm","label":"RMSNorm + Q/K/V/gate projections"},{"id":"rope","label":"Q/K RMSNorm + partial RoPE"},{"id":"attn","label":"Causal grouped-query attention"},{"id":"out","label":"Sigmoid gate / projection / residual"},{"id":"ff","label":"RMSNorm + SwiGLU"},{"id":"r2","label":"Residual add"}],"edges":[["norm","rope"],["rope","attn"],["attn","out"],["out","ff"],["ff","r2"]],"residuals":[["norm","out"],["ff","r2"]]},"index-gpt":{"title":"IndexTTS semantic AR","scope":"Both versions use GPT-style causal blocks. Version 2 has Perceiver speaker tokens; 2.5 uses a projected CAM++ vector and language embedding. Emotion and duration conditioning are separate prompt features.","blocks":[{"id":"text","label":"Text / speech embeddings + learned positions"},{"id":"cond","label":"Speaker / emotion / duration prompt"},{"id":"gpt","label":"GPT-style causal Transformer","expand":"confucius-gpt-block"},{"id":"head","label":"LayerNorm + semantic logits"}],"edges":[["text","gpt"],["cond","gpt"],["gpt","head"]]},"index-condition":{"title":"IndexTTS conditioning encoder","scope":"Speaker conditioning uses this route in version 2; emotion conditioning uses it in both versions. Independent weights implement the two branches.","blocks":[{"id":"sub","label":"Convolutional subsampling"},{"id":"enc","label":"Relative-attention Conformer"},{"id":"norm","label":"Final LayerNorm"},{"id":"query","label":"Learned latent queries"},{"id":"cross","label":"Perceiver cross-attention + feed-forward"},{"id":"out","label":"Fixed-length conditioning tokens"}],"edges":[["sub","enc"],["enc","norm"],["norm","cross"],["query","cross"],["cross","out"]]},"index-emotion":{"title":"Text-to-emotion control","scope":"An optional auxiliary Qwen3 model predicts emotion weights, which the session maps to trained emotion conditioning. Explicit weights bypass text inference.","blocks":[{"id":"tok","label":"Emotion prompt tokenization","expand":"bpe"},{"id":"qwen","label":"Qwen3 AR text generation","expand":"qwen3-layer"},{"id":"parse","label":"Parse emotion weights"},{"id":"mix","label":"Trained emotion-vector mixture"}],"edges":[["tok","qwen"],["qwen","parse"],["parse","mix"]]},"index-vq":{"title":"IndexTTS 2 semantic codes","scope":"Reference encoding and generated-code lookup meet at the same codebook. Only the generated branch adds projected AR hidden states. This semantic codec is not the final waveform vocoder.","blocks":[{"id":"ref","label":"Reference semantic ConvNeXt encoder"},{"id":"proj","label":"Projection / L2 normalization"},{"id":"assign","label":"Nearest normalized code assignment"},{"id":"generated","label":"Generated semantic IDs"},{"id":"embed","label":"Codebook embedding + output projection"},{"id":"ar","label":"Generated AR hidden-state projection"},{"id":"out","label":"Semantic features / generated feature sum"}],"edges":[["ref","proj"],["proj","assign"],["assign","embed"],["generated","embed"],["embed","out"],["ar","out"]]},"index-enhanced-codec":{"title":"IndexTTS 2.5 semantic decode","scope":"Generated codes are decoded to semantic features with 2x temporal upsampling. Reference semantic features bypass this decoder.","blocks":[{"id":"embed","label":"Codebook lookup + projection"},{"id":"conv","label":"ConvNeXt decoder backbone","expand":"convnext"},{"id":"proj","label":"Feature projection"},{"id":"up","label":"Nearest-neighbor 2x upsampling"},{"id":"out","label":"Temporal convolution"}],"edges":[["embed","conv"],["conv","proj"],["proj","up"],["up","out"]]},"index-s2mel":{"title":"IndexTTS acoustic flow","scope":"Generated and reference semantic conditions are aligned to mel length. The flow estimator combines time, CAM++ style and reference mel; BigVGAN receives only the generated mel segment.","blocks":[{"id":"align","label":"Length regulator / convolutional conditioning"},{"id":"input","label":"Noise / reference mel / style / condition projection"},{"id":"time","label":"Fourier time embeddings"},{"id":"dit","label":"Adaptive rotary DiT + long skips","expand":"confucius-dit-block"},{"id":"conv","label":"Gated convolutional refinement","expand":"confucius-wavenet"},{"id":"head","label":"Adaptive LayerNorm / mel velocity head"},{"id":"solve","label":"Flow integration / remove prompt"}],"edges":[["align","input"],["input","dit"],["time","dit"],["time","conv"],["dit","conv"],["conv","head"],["head","solve"]]},"sentencepiece":{"title":"Text to SentencePiece IDs","scope":"The checkpoint supplies vocabulary and normalization rules; the enclosing model supplies language markers and prompt structure.","blocks":[{"id":"norm","label":"Model text normalization"},{"id":"sp","label":"SentencePiece segmentation / vocabulary"},{"id":"prompt","label":"Language / special-token prompt"}],"edges":[["norm","sp"],["sp","prompt"]]},"w2v-bert":{"title":"Wav2Vec2-BERT semantic encoder","scope":"Reference audio uses normalized 80-bin filterbanks with adjacent frames stacked. Selected hidden features are normalized with the consuming model's statistics.","blocks":[{"id":"fbank","label":"Filterbanks / normalize / stack frame pairs"},{"id":"proj","label":"LayerNorm + input projection"},{"id":"enc","label":"Relative-attention Conformer stack","expand":"w2v-bert-block"},{"id":"out","label":"Select layer / feature normalization"}],"edges":[["fbank","proj"],["proj","enc"],["enc","out"]]},"w2v-bert-block":{"title":"Wav2Vec2-BERT Conformer block","scope":"Macaron feed-forward branches use half-weight residuals. Attention uses clipped relative-key embeddings. The convolution branch uses a left-padded depthwise temporal kernel.","blocks":[{"id":"ff1","label":"LayerNorm / SiLU FFN / half residual"},{"id":"attn","label":"LayerNorm / relative attention / residual"},{"id":"conv","label":"LayerNorm / pointwise GLU"},{"id":"dw","label":"Causal depthwise convolution"},{"id":"act","label":"LayerNorm / SiLU / pointwise"},{"id":"add","label":"Convolution residual add"},{"id":"ff2","label":"LayerNorm / SiLU FFN / half residual"},{"id":"norm","label":"Final LayerNorm"}],"edges":[["ff1","attn"],["attn","conv"],["conv","dw"],["dw","act"],["act","add"],["add","ff2"],["ff2","norm"]],"residuals":[["conv","add"]]},"confucius-speaker":{"title":"Semantic-feature speaker prompt","scope":"The input is a Wav2Vec2-BERT feature sequence, not raw mel. Multi-layer aggregation and attentive mean/standard-deviation pooling produce one conditioning token.","blocks":[{"id":"tdnn","label":"Initial TDNN"},{"id":"se","label":"SE-Res2Net TDNN stack"},{"id":"agg","label":"Concatenate block outputs + TDNN"},{"id":"pool","label":"Attentive statistics pooling"},{"id":"out","label":"Pointwise projection / speaker token"}],"edges":[["tdnn","se"],["se","agg"],["agg","pool"],["pool","out"]]},"confucius-t2s":{"title":"Confucius semantic AR","scope":"Text and speech have learned absolute positional embeddings. The generated sequence supplies both discrete codes and hidden states to the acoustic stage; this is not a Qwen or Llama backbone.","blocks":[{"id":"embed","label":"Text / speech embeddings + learned positions"},{"id":"prompt","label":"Speaker-token prompt"},{"id":"stack","label":"Causal GPT-style blocks","expand":"confucius-gpt-block"},{"id":"norm","label":"Final LayerNorm"},{"id":"head","label":"Semantic logits / sampling"},{"id":"hidden","label":"Hidden-state acoustic conditioning"}],"edges":[["embed","stack"],["prompt","stack"],["stack","norm"],["norm","head"],["norm","hidden"]]},"confucius-gpt-block":{"title":"GPT-style AR block","scope":"Pre-LayerNorm causal multi-head attention with packed QKV projections, followed by a GELU feed-forward residual. Positions are added outside the blocks, not through RoPE.","blocks":[{"id":"n1","label":"LayerNorm"},{"id":"attn","label":"Causal multi-head attention"},{"id":"r1","label":"Output projection + residual"},{"id":"n2","label":"LayerNorm"},{"id":"ff","label":"Linear / GELU / linear"},{"id":"r2","label":"Residual add"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"confucius-s2a":{"title":"Semantic-to-acoustic flow","scope":"The noise-to-mel solve conditions on generated semantic codes and AR latents plus reference mel and CAM++ style. Prompt mel frames are removed before vocoding.","blocks":[{"id":"codes","label":"Semantic embeddings + AR hidden projection"},{"id":"align","label":"Nearest mel-rate alignment + Conv1d"},{"id":"input","label":"Noise / reference mel / style / condition projection"},{"id":"time","label":"Fourier time embedding + MLP"},{"id":"dit","label":"Rotary DiT / long concatenative skips","expand":"confucius-dit-block"},{"id":"wavenet","label":"Gated convolutional refinement","expand":"confucius-wavenet"},{"id":"out","label":"Adaptive normalization + mel velocity head"}],"edges":[["codes","align"],["align","input"],["input","dit"],["time","dit"],["time","wavenet"],["dit","wavenet"],["wavenet","out"]]},"confucius-dit-block":{"title":"Adaptive RMS DiT block","scope":"Time-conditioned adaptive RMS normalization surrounds noncausal rotary self-attention and SwiGLU. Across the stack, later layers also receive concatenated earlier-layer skips.","blocks":[{"id":"n1","label":"Time-adaptive RMSNorm"},{"id":"attn","label":"RoPE self-attention"},{"id":"r1","label":"Residual add"},{"id":"n2","label":"Time-adaptive RMSNorm"},{"id":"ff","label":"SwiGLU feed-forward"},{"id":"r2","label":"Residual add"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"confucius-wavenet":{"title":"Gated convolutional refinement","scope":"Time-conditioned convolutions refine the DiT representation. Tanh and sigmoid gates feed residual/skip projections; this does not autoregressively synthesize waveform samples.","blocks":[{"id":"input","label":"DiT features + noise projection"},{"id":"conv","label":"Reflect-padded temporal convolution"},{"id":"time","label":"Projected time condition"},{"id":"gate","label":"Tanh x sigmoid gate"},{"id":"res","label":"Residual / skip projections"},{"id":"out","label":"Accumulate skip features"}],"edges":[["input","conv"],["conv","gate"],["time","gate"],["gate","res"],["res","out"]]},"glm-whisper-vq":{"title":"GLM reference speech tokenization","scope":"This is the speech tokenizer, not Whisper text recognition. Average pooling reduces the time axis before learned codebook scores select each discrete token.","blocks":[{"id":"mel","label":"Log-mel frontend"},{"id":"enc","label":"Whisper audio encoder layers","expand":"whisper-encoder"},{"id":"pool","label":"Temporal average pooling"},{"id":"head","label":"Codebook score projection"},{"id":"vq","label":"Argmax codebook assignment"}],"edges":[["mel","enc"],["enc","pool"],["pool","head"],["head","vq"]]},"glm-flow":{"title":"GLM acoustic flow","scope":"Reference and generated codes form a single sequence aligned to the target mel rate. The offline estimator uses noncausal rotary attention. Prompt mel frames are excluded from the final synthesis segment.","blocks":[{"id":"tokens","label":"Nearest token-to-mel alignment + embeddings"},{"id":"pos","label":"Add sinusoidal positions"},{"id":"conv","label":"ConvNeXtV2 token modeling","expand":"glm-convnext-v2"},{"id":"input","label":"Concatenate noise / reference mel + project"},{"id":"cpos","label":"Add grouped-convolution position features"},{"id":"cond","label":"Time embedding + speaker vector"},{"id":"dit","label":"Adaptive rotary DiT stack","expand":"glm-flow-block"},{"id":"out","label":"Adaptive LayerNorm + velocity projection"}],"edges":[["tokens","pos"],["pos","conv"],["conv","input"],["input","cpos"],["cpos","dit"],["cond","dit"],["dit","out"]]},"glm-convnext-v2":{"title":"ConvNeXtV2 token block","scope":"Global response normalization distinguishes this block from ordinary ConvNeXt. It models mel-aligned speech-token embeddings, not waveform samples.","blocks":[{"id":"dw","label":"Depthwise temporal convolution"},{"id":"norm","label":"LayerNorm"},{"id":"ff","label":"Linear expansion + GELU"},{"id":"grn","label":"Global response normalization"},{"id":"proj","label":"Linear contraction"},{"id":"out","label":"Residual add"}],"edges":[["dw","norm"],["norm","ff"],["ff","grn"],["grn","proj"],["proj","out"]],"residuals":[["dw","out"]]},"glm-flow-block":{"title":"GLM acoustic DiT block","scope":"Time and speaker identity jointly produce the adaptive normalization parameters. Rotary self-attention and GELU feed-forward branches have separate residual gates.","blocks":[{"id":"n1","label":"Time / speaker adaptive LayerNorm"},{"id":"attn","label":"RoPE noncausal self-attention"},{"id":"r1","label":"Gate + residual add"},{"id":"n2","label":"Adaptive LayerNorm"},{"id":"ff","label":"Linear / GELU / linear"},{"id":"r2","label":"Gate + residual add"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"s3-tokenizer":{"title":"S3 speech tokenization","scope":"The reference waveform is converted to log-mel features. A temporal attention/memory encoder and finite scalar quantizer produce one discrete content token per encoded frame.","blocks":[{"id":"mel","label":"Log-mel frontend"},{"id":"conv","label":"Conv1d / GELU subsampling"},{"id":"enc","label":"Rotary attention + FSMN stack","expand":"chatterbox-s3-block"},{"id":"proj","label":"Low-dimensional projection"},{"id":"fsq","label":"Tanh / rounding / FSQ token packing"}],"edges":[["mel","conv"],["conv","enc"],["enc","proj"],["proj","fsq"]]},"cosyvoice3-flow":{"title":"CosyVoice3 acoustic flow","scope":"Reference and generated token sequences are combined before conditioning. The flow estimates a mel sequence; prompt frames are removed before waveform synthesis.","blocks":[{"id":"codes","label":"Prompt + generated token embeddings"},{"id":"look","label":"Residual lookahead Conv1d / ReLU / Conv1d"},{"id":"up","label":"Nearest-neighbor mel-rate upsampling"},{"id":"ref","label":"Reference mel + projected speaker identity"},{"id":"noise","label":"Noise + time embedding"},{"id":"dit","label":"Conditional mel DiT","expand":"cosyvoice3-dit"},{"id":"out","label":"Flow integration / remove prompt mel"}],"edges":[["codes","look"],["look","up"],["up","dit"],["ref","dit"],["noise","dit"],["dit","out"]]},"cosyvoice3-dit":{"title":"CosyVoice3 flow Transformer","scope":"Concatenated noisy mel, reference mel, token condition and speaker condition are projected to the model width. Time conditioning supplies modulation, not an extra audio token.","blocks":[{"id":"input","label":"Concatenate conditions + linear projection"},{"id":"pos","label":"Add causal grouped-convolution position features"},{"id":"time","label":"Sinusoidal time embedding + SiLU MLP"},{"id":"stack","label":"Adaptive rotary DiT blocks","expand":"cosyvoice3-dit-block"},{"id":"out","label":"Adaptive LayerNorm + mel velocity head"}],"edges":[["input","pos"],["pos","stack"],["time","stack"],["stack","out"]]},"cosyvoice3-dit-block":{"title":"CosyVoice3 DiT block","scope":"The offline path uses noncausal rotary attention. Time-conditioned shift, scale and gates modulate the attention and GELU feed-forward residual branches.","blocks":[{"id":"n1","label":"Adaptive LayerNorm"},{"id":"attn","label":"RoPE noncausal self-attention"},{"id":"r1","label":"Gate + residual add"},{"id":"n2","label":"Adaptive LayerNorm"},{"id":"ff","label":"Linear / GELU / linear"},{"id":"r2","label":"Gate + residual add"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"redae-encoder":{"title":"RedAE continuous audio encoding","scope":"A waveform patch projection feeds sliding-window causal attention. Local groups append a learned CLS token; another causal Transformer produces the compressed representation. There is no codebook quantization.","blocks":[{"id":"patch","label":"Waveform patches + linear projections"},{"id":"tr","label":"Sliding-window Qwen3-style stack","expand":"qwen3-layer"},{"id":"group","label":"Group adjacent frames + append CLS"},{"id":"down","label":"Local Qwen3-style downsampler","expand":"qwen3-layer"},{"id":"out","label":"Select CLS + bottleneck projection"}],"edges":[["patch","tr"],["tr","group"],["group","down"],["down","out"]]},"redae-decoder":{"title":"RedAE waveform reconstruction","scope":"The latent projection expands the temporal rate before the causal decoder. The output head predicts log magnitude and phase, not waveform samples or discrete tokens directly.","blocks":[{"id":"proj","label":"Linear projection + temporal unpacking"},{"id":"tr","label":"Sliding-window Qwen3-style stack","expand":"qwen3-layer"},{"id":"head","label":"Linear log-magnitude / phase head"},{"id":"spec","label":"Complex spectrum construction"},{"id":"istft","label":"ISTFT + overlap-add"}],"edges":[["proj","tr"],["tr","head"],["head","spec"],["spec","istft"]]},"firered-patch":{"title":"Latent patch summarization","scope":"Each acoustic patch is independently summarized. The same encoder embeds reference patches and feeds generated patches back into the next Qwen3 AR step; the pipeline omits that recurrent edge for readability.","blocks":[{"id":"proj","label":"Latent-frame projection"},{"id":"cls","label":"Prepend learned CLS token"},{"id":"tr","label":"Noncausal rotary Transformer","expand":"firered-patch-block"},{"id":"norm","label":"RMSNorm + output projection"},{"id":"select","label":"Select CLS patch embedding"}],"edges":[["proj","cls"],["cls","tr"],["tr","norm"],["norm","select"]]},"firered-patch-block":{"title":"FireRed latent-patch block","scope":"Rotary self-attention is bidirectional within a patch. The block uses RMS normalization and a GELU feed-forward, distinct from the Qwen3 backbone's SwiGLU.","blocks":[{"id":"n1","label":"RMSNorm"},{"id":"attn","label":"RoPE noncausal multi-head attention"},{"id":"r1","label":"Residual add"},{"id":"n2","label":"RMSNorm"},{"id":"ff","label":"Linear / GELU / linear"},{"id":"r2","label":"Residual add"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"firered-flow":{"title":"FireRed patch flow","scope":"History patches stay fixed while the current noise patch is integrated. Qwen3 states condition every patch position; Base also concatenates projected speaker identity. The stop head on the AR state terminates generation.","blocks":[{"id":"history","label":"History latents + current noise patch"},{"id":"cond","label":"Projected AR states / Base speaker vector"},{"id":"input","label":"Concatenate channels + input projection"},{"id":"time","label":"Sinusoidal time embedding + MLP"},{"id":"stack","label":"Time-modulated DiT blocks","expand":"firered-flow-block"},{"id":"out","label":"Modulated LayerNorm + velocity head"},{"id":"solve","label":"Integrate current patch"}],"edges":[["history","input"],["cond","input"],["input","stack"],["time","stack"],["stack","out"],["out","solve"]]},"firered-flow-block":{"title":"FireRed acoustic DiT block","scope":"Time conditioning predicts shift, scale and gate parameters for all three residual branches. Attention mixes patch positions, convolution models local detail, and the feed-forward is GELU based.","blocks":[{"id":"n1","label":"Adaptive RMSNorm"},{"id":"attn","label":"RoPE noncausal self-attention"},{"id":"r1","label":"Gate + residual add"},{"id":"n2","label":"Adaptive RMSNorm"},{"id":"conv","label":"Conv1d / Mish / Conv1d"},{"id":"r2","label":"Gate + residual add"},{"id":"n3","label":"Adaptive RMSNorm"},{"id":"ff","label":"Linear / GELU / linear"},{"id":"r3","label":"Gate + residual add"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","conv"],["conv","r2"],["r2","n3"],["n3","ff"],["ff","r3"]],"residuals":[["n1","r1"],["n2","r2"],["n3","r3"]]},"vieneu-frontend":{"title":"VieNeu phoneme frontend","scope":"SEA-G2P is optional in AudioCPP. Without its configured dictionary, the input must already contain phonemes; there is no automatic generic English tokenizer substitution.","blocks":[{"id":"g2p","label":"Optional SEA-G2P text-to-phones"},{"id":"tokens","label":"Phoneme vocabulary encoding"},{"id":"prompt","label":"Prompt boundary / control tokens"}],"edges":[["g2p","tokens"],["tokens","prompt"]]},"vieneu-anchor":{"title":"Speaker anchor","scope":"Packaged presets or supplied embeddings provide the speaker vector. The current model package does not include upstream CAM++ extraction; this projection is not a speaker recognition network.","blocks":[{"id":"vec","label":"192-dimensional speaker vector"},{"id":"linear","label":"Linear projection"},{"id":"norm","label":"LayerNorm"},{"id":"add","label":"Add anchor to prompt embeddings"}],"edges":[["vec","linear"],["linear","norm"],["norm","add"]]},"vieneu-talker":{"title":"VieNeu temporal AR","scope":"Text and reference-code frames form the prompt. Each generated frame's summed embeddings feed the next temporal step. The talker is trained from scratch, with Qwen3-style blocks rather than pretrained Qwen weights.","blocks":[{"id":"embed","label":"Phone / summed codebook embeddings"},{"id":"anchor","label":"Add speaker anchor"},{"id":"stack","label":"Rotary causal Transformer stack","expand":"vieneu-temporal-block"},{"id":"norm","label":"RMSNorm"},{"id":"out","label":"Temporal state for local decoder"}],"edges":[["embed","anchor"],["anchor","stack"],["stack","norm"],["norm","out"]]},"vieneu-temporal-block":{"title":"VieNeu temporal Transformer block","scope":"Pre-normalized causal attention uses rotary positions and normalized Q/K. The feed-forward branch is SiLU-gated.","blocks":[{"id":"n1","label":"RMSNorm"},{"id":"attn","label":"Q/K norm + RoPE causal attention"},{"id":"r1","label":"Residual add"},{"id":"n2","label":"RMSNorm"},{"id":"ff","label":"SwiGLU"},{"id":"r2","label":"Residual add"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"vieneu-depth":{"title":"VieNeu within-frame AR","scope":"The local decoder resets for each frame. Learned slot positions distinguish the temporal state and codebook tokens. Its control head and audio heads have separate vocabularies.","blocks":[{"id":"embed","label":"Temporal state + previous code embedding"},{"id":"pos","label":"Add learned slot positions"},{"id":"stack","label":"Local causal Transformer","expand":"vieneu-depth-block"},{"id":"norm","label":"RMSNorm"},{"id":"control","label":"Text / control head"},{"id":"audio","label":"Per-codebook audio heads"}],"edges":[["embed","pos"],["pos","stack"],["stack","norm"],["norm","control"],["norm","audio"]]},"vieneu-depth-block":{"title":"VieNeu acoustic Transformer block","scope":"Unlike the temporal block, this decoder uses learned slot embeddings and no rotary positions. Packed QKV projection feeds Q/K-normalized causal multi-head attention.","blocks":[{"id":"n1","label":"RMSNorm"},{"id":"attn","label":"Packed QKV + Q/K norm + causal MHA"},{"id":"r1","label":"Residual add"},{"id":"n2","label":"RMSNorm"},{"id":"ff","label":"SwiGLU"},{"id":"r2","label":"Residual add"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"audio8-slow":{"title":"Audio8 semantic AR","scope":"The packaged 0.6B model uses a 24-layer rotary GQA backbone. Summed codebook embeddings feed the next temporal step; reference codes and transcripts occupy the initial prompt.","blocks":[{"id":"embed","label":"Text + summed codebook embeddings"},{"id":"tr","label":"Causal GQA Transformer stack","expand":"audio8-ar-block"},{"id":"norm","label":"Final RMSNorm"},{"id":"head","label":"Semantic token head"},{"id":"hidden","label":"Hidden state for depth decoder"}],"edges":[["embed","tr"],["tr","norm"],["norm","head"],["norm","hidden"]]},"audio8-fast":{"title":"Audio8 within-frame AR","scope":"The four-layer fast decoder uses the slow state and semantic token to predict the rest of the ten-codebook frame. Unlike slow 0.6B attention, its QKV projections are bias-free.","blocks":[{"id":"in","label":"Slow state + first-code conditioning"},{"id":"prev","label":"Previous code embedding"},{"id":"tr","label":"Causal depth Transformer stack","expand":"audio8-ar-block"},{"id":"head","label":"RMSNorm + audio logits"}],"edges":[["in","tr"],["prev","tr"],["tr","head"]]},"audio8-ar-block":{"title":"Audio8 rotary GQA block","scope":"Both branches use RMS normalization, rotary grouped-query attention without Q/K normalization, and a SiLU-gated feed-forward sublayer.","blocks":[{"id":"n1","label":"RMSNorm"},{"id":"attn","label":"RoPE causal grouped-query attention"},{"id":"r1","label":"Residual add"},{"id":"n2","label":"RMSNorm"},{"id":"ff","label":"SwiGLU feed-forward"},{"id":"r2","label":"Residual add"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"audio8-falcon":{"title":"Alternate Falcon-H1 slow backbone","scope":"This depicts the alternate implementation in AudioCPP, not an additional downloadable package in its spec. Each layer combines parallel recurrent and attention branches before the feed-forward update.","blocks":[{"id":"in","label":"Scaled text / audio embeddings"},{"id":"tr","label":"Hybrid decoder layers","expand":"audio8-falcon-block"},{"id":"out","label":"RMSNorm + semantic head"}],"edges":[["in","tr"],["tr","out"]]},"audio8-falcon-block":{"title":"Falcon-H1 parallel hybrid block","scope":"Normalized input feeds both branches. Mamba2 maintains convolution and SSM state; rotary attention maintains K/V state. The projected branch outputs join the residual stream.","blocks":[{"id":"norm","label":"RMSNorm"},{"id":"ssm","label":"Mamba2 recurrent mixer","expand":"audio8-mamba2"},{"id":"attn","label":"Rotary grouped-query attention"},{"id":"join","label":"Sum branches + residual"},{"id":"ff","label":"RMSNorm + SwiGLU"},{"id":"out","label":"Residual add"}],"edges":[["norm","ssm"],["norm","attn"],["ssm","join"],["attn","join"],["join","ff"],["ff","out"]],"residuals":[["norm","join"],["ff","out"]]},"audio8-mamba2":{"title":"Mamba2 recurrent mixer","scope":"The single-step path updates cached convolution and state-space states. Input-dependent B/C and time steps govern the recurrent update; the gate modulates its output.","blocks":[{"id":"proj","label":"Project gate / x / B / C / time step"},{"id":"conv","label":"Cached depthwise convolution + SiLU"},{"id":"scan","label":"Mamba2 state update + skip term"},{"id":"gate","label":"Multiply by SiLU gate"},{"id":"out","label":"Output projection"}],"edges":[["proj","conv"],["conv","scan"],["scan","gate"],["gate","out"]]},"audio8-codec":{"title":"Audio8 waveform reconstruction","scope":"Semantic and residual embeddings are projected and summed. Audio8's codec uses its own grouped-query Transformer dimensions and codebooks; the high-level stages follow the Fish-derived codec design.","blocks":[{"id":"rvq","label":"Codebook lookup / projected residual sum"},{"id":"tr","label":"Windowed rotary Transformer","expand":"fish-codec-block"},{"id":"up","label":"ConvNeXt / transposed-convolution rate conversion","expand":"midasheng-convnext"},{"id":"wave","label":"Snake residual waveform upsampling","expand":"sam-dac-residual"},{"id":"out","label":"Waveform convolution + tanh"}],"edges":[["rvq","tr"],["tr","up"],["up","wave"],["wave","out"]]},"fish-slow":{"title":"Fish slow temporal AR","scope":"The decoder predicts the primary semantic codebook along time. Generated codebook embeddings are summed into the next frame's input; reference codes occupy prompt spans alongside their transcripts.","blocks":[{"id":"embed","label":"Text / semantic + summed audio embeddings"},{"id":"tr","label":"Causal rotary decoder stack","expand":"fish-ar-block"},{"id":"norm","label":"Final RMSNorm"},{"id":"head","label":"Semantic logits + sampling"},{"id":"hidden","label":"Hidden state for fast decoder"}],"edges":[["embed","tr"],["tr","norm"],["norm","head"],["norm","hidden"]]},"fish-fast":{"title":"Fish fast depth AR","scope":"The slow hidden state seeds a fresh short sequence per frame. The sampled semantic code is the first audio code; subsequent depth steps predict residual codes.","blocks":[{"id":"seed","label":"Slow hidden-state prefix"},{"id":"embed","label":"Semantic / preceding code embedding"},{"id":"tr","label":"Causal depth decoder stack","expand":"fish-ar-block"},{"id":"head","label":"RMSNorm + codebook logits"},{"id":"out","label":"Complete frame / global feedback"}],"edges":[["seed","tr"],["embed","tr"],["tr","head"],["head","out"]]},"fish-ar-block":{"title":"Fish rotary AR block","scope":"Slow and fast decoders have separate parameters. Both use RMSNorm, rotary causal attention and SwiGLU; Q/K normalization follows the checkpoint configuration.","blocks":[{"id":"n1","label":"RMSNorm"},{"id":"attn","label":"Q/K norm if configured / RoPE causal attention"},{"id":"r1","label":"Residual add"},{"id":"n2","label":"RMSNorm"},{"id":"ff","label":"SwiGLU feed-forward"},{"id":"r2","label":"Residual add"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"fish-dac-codes":{"title":"Fish discrete-code waveform synthesis","scope":"Unlike Echo's latent entry point, Fish first looks up generated code IDs. The reconstructed latent then enters the same high-level Transformer and convolutional synthesis architecture.","blocks":[{"id":"lookup","label":"Codebook lookup + per-codebook projection"},{"id":"sum","label":"Sum semantic / residual contributions"},{"id":"tr","label":"Windowed post-quantizer Transformer","expand":"fish-codec-block"},{"id":"up","label":"Transposed convolution + ConvNeXt rate conversion","expand":"midasheng-convnext"},{"id":"wave","label":"Snake upsampling waveform CNN","expand":"music3-vae-residual"},{"id":"out","label":"Waveform convolution + tanh"}],"edges":[["lookup","sum"],["sum","tr"],["tr","up"],["up","wave"],["wave","out"]]},"fish-codec-block":{"title":"Fish codec windowed Transformer block","scope":"The codec uses window-limited attention with rotary positions, RMSNorm, SwiGLU and LayerScale. It is distinct from the text/audio AR decoders.","blocks":[{"id":"n1","label":"RMSNorm"},{"id":"attn","label":"Windowed rotary attention"},{"id":"r1","label":"LayerScale + residual"},{"id":"n2","label":"RMSNorm"},{"id":"ff","label":"SwiGLU feed-forward"},{"id":"r2","label":"LayerScale + residual"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"moss-local-depth":{"title":"MOSS Local frame generation","scope":"Each frame starts from a Qwen3 hidden state. A binary control head decides continuation; codebooks are sampled sequentially and their summed embeddings feed the next global AR step.","blocks":[{"id":"hidden","label":"Global Qwen3 hidden state"},{"id":"codes","label":"Earlier codebook embeddings in this frame"},{"id":"tr","label":"Local causal rotary Transformer","expand":"moss-local-block"},{"id":"control","label":"Binary continuation head"},{"id":"audio","label":"Codebook projection + sampling"}],"edges":[["hidden","tr"],["codes","tr"],["tr","control"],["tr","audio"]]},"moss-local-block":{"title":"MOSS Local depth block","scope":"LayerNorm precedes rotary attention and a plain SiLU feed-forward. This is not the Qwen3 RMSNorm / SwiGLU block used by the global backbone.","blocks":[{"id":"n1","label":"LayerNorm"},{"id":"attn","label":"RoPE causal multi-head attention"},{"id":"r1","label":"Residual add"},{"id":"n2","label":"LayerNorm"},{"id":"ff","label":"Linear / SiLU / linear"},{"id":"r2","label":"Residual add"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"moss-nano-global":{"title":"MOSS Nano temporal backbone","scope":"Text embeddings and available audio-codebook embeddings are summed per position. Global states seed a separate within-frame decoder; Nano is not a Qwen3 derivative.","blocks":[{"id":"embed","label":"Text + summed audio-codebook embeddings"},{"id":"tr","label":"Causal rotary Transformer stack","expand":"moss-nano-block"},{"id":"norm","label":"Final LayerNorm"},{"id":"out","label":"Global frame state"}],"edges":[["embed","tr"],["tr","norm"],["norm","out"]]},"moss-nano-depth":{"title":"MOSS Nano local frame decoder","scope":"The local decoder first predicts a text/control token from the global state, then uses that token and earlier codebooks to predict each audio code. A new short depth sequence starts for every global step.","blocks":[{"id":"global","label":"Global hidden state"},{"id":"embed","label":"Text/control token + earlier audio embeddings"},{"id":"tr","label":"Causal local Transformer stack","expand":"moss-nano-block"},{"id":"norm","label":"Final LayerNorm"},{"id":"head","label":"Text or codebook-specific linear head"}],"edges":[["global","tr"],["embed","tr"],["tr","norm"],["norm","head"]]},"moss-nano-block":{"title":"MOSS Nano rotary AR block","scope":"The global and local decoders share this block structure but have separate weights and dimensions. Attention is causal with rotary positions, followed by a GELU feed-forward.","blocks":[{"id":"n1","label":"LayerNorm"},{"id":"attn","label":"RoPE causal multi-head attention"},{"id":"r1","label":"Residual add"},{"id":"n2","label":"LayerNorm"},{"id":"ff","label":"Linear / GELU / linear"},{"id":"r2","label":"Residual add"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"moss-codec-encoder":{"title":"MOSS reference audio tokenization","scope":"Each encoder stage patches the temporal sequence before its projected Transformer. The quantizer repeatedly encodes the residual with normalized codebook matching.","blocks":[{"id":"patch","label":"Waveform patches / channel packing"},{"id":"stages","label":"Patch / projected causal Transformer stages","expand":"moss-codec-block"},{"id":"proj","label":"Quantizer input projection"},{"id":"rvq","label":"Residual quantization / nearest code matching"},{"id":"out","label":"Reference codebooks"}],"edges":[["patch","stages"],["stages","proj"],["proj","rvq"],["rvq","out"]]},"moss-codec-decoder":{"title":"MOSS Transformer waveform decoding","scope":"The decoder uses temporal unpatching after each projected Transformer stage. It reconstructs waveform samples directly, without a mel vocoder or ISTFT.","blocks":[{"id":"codes","label":"Codebook lookup + per-codebook projection"},{"id":"sum","label":"Sum residual contributions + output projection"},{"id":"tr","label":"Projected causal Transformer stages","expand":"moss-codec-block"},{"id":"patch","label":"Temporal unpatching at each stage"},{"id":"out","label":"Waveform / channel unpacking"}],"edges":[["codes","sum"],["sum","tr"],["tr","patch"],["patch","out"]]},"moss-codec-block":{"title":"Causal audio Transformer block","scope":"Each stage has its own local causal attention window and temporal rate. LayerScale gates both residual updates; stage projections change width when needed.","blocks":[{"id":"n1","label":"LayerNorm"},{"id":"attn","label":"RoPE / local causal self-attention"},{"id":"r1","label":"LayerScale + residual"},{"id":"n2","label":"LayerNorm"},{"id":"ff","label":"Linear / GELU / linear"},{"id":"r2","label":"LayerScale + residual"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"moss-delay-heads":{"title":"Delayed multi-codebook prediction","scope":"At each AR step, one hidden vector feeds separate output heads. Delays allow codebooks to depend on earlier codebooks through subsequent AR steps. Generated row embeddings are summed for backbone feedback.","blocks":[{"id":"hidden","label":"AR hidden state"},{"id":"text","label":"Text / control-token linear head"},{"id":"audio","label":"Parallel audio-codebook linear heads"},{"id":"schedule","label":"Delay-pattern sampling / flush"},{"id":"align","label":"Undo codebook delays / aligned frames"}],"edges":[["hidden","text"],["hidden","audio"],["text","schedule"],["audio","schedule"],["schedule","align"]]},"music3-depth":{"title":"Within-frame codebook generation","scope":"Global hidden conditioning and the sampled semantic code seed a short causal sequence. Seven codebook-specific heads generate residual codes. Their embeddings are summed with the semantic embedding for global-model feedback on the next frame.","blocks":[{"id":"in","label":"Project global hidden / semantic embedding"},{"id":"embed","label":"Residual-code embeddings + depth positions"},{"id":"tr","label":"Local causal Transformer","expand":"music3-depth-block"},{"id":"head","label":"RMSNorm + per-codebook logits"},{"id":"states","label":"Preserve local hidden states for synthesis"}],"edges":[["in","tr"],["embed","tr"],["tr","head"],["tr","states"]]},"music3-depth-block":{"title":"Local depth decoder block","scope":"Learned positions are added before the stack. The attention sublayer uses no RoPE or Q/K normalization; the feed-forward is gated SiLU.","blocks":[{"id":"n1","label":"RMSNorm"},{"id":"attn","label":"Causal multi-head attention"},{"id":"r1","label":"Residual add"},{"id":"n2","label":"RMSNorm"},{"id":"ff","label":"SwiGLU feed-forward"},{"id":"r2","label":"Residual add"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"music3-fusion":{"title":"Hidden-state fusion","scope":"Synthesis consumes continuous global and local decoder states, not codebook IDs. This conditioner is a weighted mixture and convolution, not another text encoder.","blocks":[{"id":"states","label":"Global + local frame hidden states"},{"id":"mix","label":"Learned softmax mixture + scale"},{"id":"conv","label":"Temporal convolution"},{"id":"rate","label":"Nearest-neighbor latent-rate alignment"}],"edges":[["states","mix"],["mix","conv"],["conv","rate"]]},"music3-flow":{"title":"Music latent flow synthesis","scope":"The denoiser combines noise, a zero-filled conditioning slot and fused AR features. Fourier time features become a prepended token rather than adaptive normalization parameters.","blocks":[{"id":"cat","label":"Concatenate noise / zero slot / AR features"},{"id":"proj","label":"Residual pointwise convolution + projection"},{"id":"time","label":"Fourier time embedding + SiLU MLP"},{"id":"join","label":"Prepend time token"},{"id":"tr","label":"Noncausal flow Transformer stack","expand":"music3-flow-block"},{"id":"out","label":"Remove time token + latent velocity head"}],"edges":[["cat","proj"],["proj","join"],["time","join"],["join","tr"],["tr","out"]]},"music3-flow-block":{"title":"Music flow Transformer block","scope":"Attention uses partial rotary position encoding. Pre-LayerNorm residual attention and SwiGLU layers refine all latent frames together.","blocks":[{"id":"n1","label":"LayerNorm"},{"id":"attn","label":"Partial-RoPE noncausal self-attention"},{"id":"r1","label":"Residual add"},{"id":"n2","label":"LayerNorm"},{"id":"ff","label":"SwiGLU feed-forward"},{"id":"r2","label":"Residual add"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"music3-vae":{"title":"Stereo Flow-VAE decoding","scope":"The two channels are decoded through the convolutional waveform path and interleaved. Continuous flow latents replace any discrete-code lookup.","blocks":[{"id":"proj","label":"Latent projection + input convolution"},{"id":"up","label":"Snake + transposed convolution stages"},{"id":"res","label":"Dilated Snake residual units","expand":"music3-vae-residual"},{"id":"out","label":"Snake / output convolution / tanh"},{"id":"stereo","label":"Interleave stereo channels"}],"edges":[["proj","up"],["up","res"],["res","out"],["out","stereo"]]},"music3-vae-residual":{"title":"Waveform residual unit","scope":"Each upsampling stage contains temporal residual units with different dilations.","blocks":[{"id":"s1","label":"Snake activation"},{"id":"conv","label":"Dilated temporal convolution"},{"id":"s2","label":"Snake + pointwise convolution"},{"id":"out","label":"Residual add"}],"edges":[["s1","conv"],["conv","s2"],["s2","out"]],"residuals":[["s1","out"]]},"midasheng-ar":{"title":"Continuous-patch Qwen3 generation","scope":"After each flow solve, the generated patch is flattened and projected through a GELU MLP into the next Qwen input embedding. The diagram shows one step; the stop probabilities trim the final waveform.","blocks":[{"id":"text","label":"Text token embedding lookup"},{"id":"patch","label":"Previous generated patch + GELU projector"},{"id":"ar","label":"Cached Qwen3 causal blocks","expand":"qwen3-layer"},{"id":"hidden","label":"Final hidden state / flow conditioning"},{"id":"stop","label":"Linear two-class stop head"}],"edges":[["text","ar"],["patch","ar"],["ar","hidden"],["hidden","stop"]]},"midasheng-flow":{"title":"Per-patch flow generation","scope":"The previous patch starts at zero. Time embedding is added to projected AR conditioning. Output selection keeps only the current noisy-patch positions.","blocks":[{"id":"cond","label":"Sinusoidal time MLP + AR projection"},{"id":"patch","label":"Project history and noisy patch"},{"id":"cat","label":"Concatenate condition / history / current tokens"},{"id":"stack","label":"Rotary Transformer stack","expand":"midasheng-flow-block"},{"id":"out","label":"RMSNorm + linear / current-patch velocity"}],"edges":[["cond","cat"],["patch","cat"],["cat","stack"],["stack","out"]]},"midasheng-flow-block":{"title":"Patch-flow Transformer block","scope":"Flow self-attention is noncausal. Unlike the Qwen3 AR backbone, this block uses a plain GELU feed-forward, not SwiGLU.","blocks":[{"id":"norm","label":"RMSNorm"},{"id":"attn","label":"Rotary self-attention"},{"id":"add","label":"Residual add"},{"id":"norm2","label":"RMSNorm"},{"id":"ff","label":"Linear / GELU / linear"},{"id":"out","label":"Residual add"}],"edges":[["norm","attn"],["attn","add"],["add","norm2"],["norm2","ff"],["ff","out"]],"residuals":[["norm","add"],["norm2","out"]]},"midasheng-codec":{"title":"Continuous audio latent decoding","scope":"Upsampling precedes the Vocos-style backbone. The spectrum head predicts magnitude and phase rather than discrete audio tokens.","blocks":[{"id":"up","label":"Transposed convolution / temporal upsampling"},{"id":"in","label":"Input convolution + LayerNorm"},{"id":"stack","label":"ConvNeXt block stack","expand":"midasheng-convnext"},{"id":"head","label":"LayerNorm + spectrum projection"},{"id":"out","label":"Magnitude / phase + ISTFT"}],"edges":[["up","in"],["in","stack"],["stack","head"],["head","out"]]},"midasheng-convnext":{"title":"Audio ConvNeXt block","scope":"Depthwise temporal convolution mixes adjacent frames. Channel expansion, GELU and LayerScale form the residual update.","blocks":[{"id":"conv","label":"Depthwise temporal convolution"},{"id":"norm","label":"LayerNorm"},{"id":"ff","label":"Linear expansion / GELU / projection"},{"id":"scale","label":"LayerScale + residual"}],"edges":[["conv","norm"],["norm","ff"],["ff","scale"]],"residuals":[["conv","scale"]]},"irodori-condition":{"title":"Irodori v3 text conditioning","scope":"Text and caption use distinct embeddings and encoders. The caption branch is available only on caption-conditioned checkpoints.","blocks":[{"id":"embed","label":"Checkpoint tokenizer + learned embeddings"},{"id":"stack","label":"Gated rotary encoder stack","expand":"irodori-condition-block"},{"id":"out","label":"RMSNorm + valid-token mask"}],"edges":[["embed","stack"],["stack","out"]]},"irodori-condition-block":{"title":"Gated rotary conditioning block","scope":"Bidirectional attention uses Q/K RMS normalization and an output sigmoid gate. A pre-normalized SwiGLU sublayer follows.","blocks":[{"id":"n1","label":"RMSNorm"},{"id":"attn","label":"Q/K norm + rotary self-attention"},{"id":"gate","label":"Sigmoid output gate + residual"},{"id":"n2","label":"RMSNorm"},{"id":"ff","label":"SwiGLU feed-forward"},{"id":"out","label":"Residual add"}],"edges":[["n1","attn"],["attn","gate"],["gate","n2"],["n2","ff"],["ff","out"]],"residuals":[["n1","gate"],["n2","out"]]},"irodori-modernbert":{"title":"ModernBERT-JA conditioning","scope":"The shared pretrained backbone encodes text and caption sequences separately. Each has its own output projector and normalization.","blocks":[{"id":"embed","label":"Token embeddings + LayerNorm"},{"id":"stack","label":"Local / global ModernBERT blocks","expand":"irodori-modernbert-block"},{"id":"norm","label":"Final LayerNorm"},{"id":"proj","label":"Text or caption projector + RMSNorm"}],"edges":[["embed","stack"],["stack","norm"],["norm","proj"]]},"irodori-modernbert-block":{"title":"ModernBERT encoder block","scope":"Bidirectional rotary attention alternates sliding-window and global layers. The feed-forward uses gated GELU rather than Irodori v3's SwiGLU.","blocks":[{"id":"n1","label":"LayerNorm"},{"id":"attn","label":"Local / global rotary self-attention"},{"id":"r1","label":"Residual add"},{"id":"n2","label":"LayerNorm"},{"id":"ff","label":"Gated GELU feed-forward"},{"id":"r2","label":"Residual add"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"irodori-speaker":{"title":"Reference-latent conditioning","scope":"The reference waveform first passes through DAC-VAE encoding. This Transformer consumes continuous latent patches, not mel features or discrete codec IDs.","blocks":[{"id":"patch","label":"Patch reference latents"},{"id":"proj","label":"Linear projection + scaling"},{"id":"tr","label":"Gated rotary Transformer stack","expand":"irodori-condition-block"},{"id":"out","label":"RMSNorm / masked speaker states"}],"edges":[["patch","proj"],["proj","tr"],["tr","out"]]},"irodori-duration":{"title":"Learned duration prediction","scope":"Text-token states are modulated by speaker and optional caption context. The estimate controls the generated latent sequence length; explicit duration bypasses its use.","blocks":[{"id":"text","label":"Project text states"},{"id":"cond","label":"Speaker / caption summary or learned nulls"},{"id":"blocks","label":"Conditional normalized token blocks"},{"id":"head","label":"Token duration projection"},{"id":"out","label":"Masked token aggregation / output length"}],"edges":[["text","blocks"],["cond","blocks"],["blocks","head"],["head","out"]]},"irodori-flow":{"title":"Irodori rectified-flow DiT","scope":"Noisy latent queries jointly attend to latent, text, speaker and caption keys/values. Half of the latent attention heads receive rotary positions. This is not autoregressive generation.","blocks":[{"id":"in","label":"Latent input projection"},{"id":"time","label":"Time embedding + low-rank modulation"},{"id":"ctx","label":"Text / speaker / caption K-V projections"},{"id":"tr","label":"Adaptive gated DiT stack","expand":"irodori-flow-block"},{"id":"out","label":"Output normalization + latent velocity"}],"edges":[["in","tr"],["time","tr"],["ctx","tr"],["tr","out"]]},"irodori-flow-block":{"title":"Irodori RF-DiT block","scope":"Low-rank time modulation supplies shifts, scales and residual gates. Attention has an additional sigmoid output gate; conditioning states supply keys and values.","blocks":[{"id":"n1","label":"Time-adaptive RMSNorm"},{"id":"attn","label":"Q/K norm + partial-RoPE joint attention"},{"id":"r1","label":"Attention output gate + gated residual"},{"id":"n2","label":"Time-adaptive RMSNorm"},{"id":"ff","label":"SwiGLU feed-forward"},{"id":"r2","label":"Gated residual"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"irodori-codec":{"title":"Continuous DAC-VAE decoding","scope":"A latent projection replaces discrete codebook lookup. Convolutional stages reconstruct the waveform directly, without a mel spectrogram or ISTFT.","blocks":[{"id":"proj","label":"Latent projection + input convolution"},{"id":"up","label":"Snake + transposed convolution stages"},{"id":"res","label":"Dilated Snake residual units"},{"id":"out","label":"Output convolution / waveform"}],"edges":[["proj","up"],["up","res"],["res","out"]]},"chatterbox-voice":{"title":"LSTM voice embedding","scope":"The T3 speaker embedding is independent of the CAMPPlus embedding used by the acoustic generator. Partial-window embeddings are averaged and normalized.","blocks":[{"id":"mel","label":"Mel frontend + overlapping windows"},{"id":"rnn","label":"Three stacked LSTM layers"},{"id":"proj","label":"Final hidden state + linear projection"},{"id":"norm","label":"L2-normalize window embeddings"},{"id":"out","label":"Average windows + L2 normalization"}],"edges":[["mel","rnn"],["rnn","proj"],["proj","norm"],["norm","out"]]},"chatterbox-s3":{"title":"S3 speech tokenization","scope":"The tokenizer produces content codes for reference prompting or voice conversion. Finite scalar quantization uses eight three-level dimensions, packed into one token ID per frame.","blocks":[{"id":"mel","label":"Log-mel frontend"},{"id":"conv","label":"Conv1D / GELU subsampling"},{"id":"enc","label":"Rotary attention + FSMN encoder","expand":"chatterbox-s3-block"},{"id":"proj","label":"Low-dimensional projection"},{"id":"fsq","label":"Tanh / rounding / FSQ token packing"}],"edges":[["mel","conv"],["conv","enc"],["enc","proj"],["proj","fsq"]]},"chatterbox-s3-block":{"title":"S3 attention / memory block","scope":"The value projection also feeds a depthwise finite-memory convolution. Its output joins the rotary attention result before the feed-forward sublayer.","blocks":[{"id":"norm","label":"LayerNorm + Q/K/V projections"},{"id":"attn","label":"Rotary self-attention"},{"id":"fsmn","label":"Value + depthwise temporal memory"},{"id":"add","label":"Add attention / memory / residual"},{"id":"ff","label":"LayerNorm + GELU feed-forward"},{"id":"out","label":"Residual add"}],"edges":[["norm","attn"],["norm","fsmn"],["attn","add"],["fsmn","add"],["add","ff"],["ff","out"]],"residuals":[["norm","add"],["ff","out"]]},"chatterbox-perceiver":{"title":"Speech-prompt resampling","scope":"Learned queries attend to normalized reference speech embeddings, then refine the result with self-attention. Speaker and emotion embeddings are separate T3 conditions.","blocks":[{"id":"embed","label":"Speech-code embedding + learned positions"},{"id":"query","label":"Learned prompt queries"},{"id":"cross","label":"Normalized query-to-prompt attention"},{"id":"self","label":"Normalized self-attention"},{"id":"out","label":"Fixed-length T3 prompt states"}],"edges":[["embed","cross"],["query","cross"],["cross","self"],["self","out"]]},"gpt2-ar":{"title":"GPT-2 autoregressive decoder","scope":"Learned absolute position embeddings cover the combined conditioning, text and speech sequence. Speech-code prediction uses a model-specific output head.","blocks":[{"id":"embed","label":"Token / condition embeddings + positions"},{"id":"stack","label":"Causal GPT-2 block stack","expand":"gpt2-block"},{"id":"norm","label":"Final LayerNorm"},{"id":"head","label":"Speech-code logits + sampling"}],"edges":[["embed","stack"],["stack","norm"],["norm","head"]]},"gpt2-block":{"title":"GPT-2 decoder block","scope":"LayerNorm and GELU distinguish this block from the RMSNorm / SwiGLU Llama backbone in regular Chatterbox.","blocks":[{"id":"n1","label":"LayerNorm"},{"id":"attn","label":"Causal multi-head attention"},{"id":"r1","label":"Residual add"},{"id":"n2","label":"LayerNorm"},{"id":"ff","label":"Linear / GELU / linear"},{"id":"r2","label":"Residual add"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"s3gen":{"title":"S3Gen acoustic generation","scope":"S3Gen maps prompt and generated speech tokens to mel features. Speaker identity and reference mel features condition the flow solve. HiFT is a separate waveform stage.","blocks":[{"id":"embed","label":"Prompt + generated token embeddings"},{"id":"enc","label":"Relative-attention token encoder","expand":"s3gen-encoder"},{"id":"up","label":"Temporal upsampling + second encoder stack"},{"id":"mu","label":"Mel-conditioning projection"},{"id":"ref","label":"Reference mel + speaker embedding"},{"id":"flow","label":"Conditional flow estimator","expand":"s3gen-estimator"},{"id":"mel","label":"Remove prompt frames / generated mel"}],"edges":[["embed","enc"],["enc","up"],["up","mu"],["mu","flow"],["ref","flow"],["flow","mel"]]},"s3gen-encoder":{"title":"S3Gen token encoder block","scope":"Pre-normalized relative-position self-attention and a SiLU feed-forward encode speech-token frames before and after upsampling.","blocks":[{"id":"n1","label":"LayerNorm"},{"id":"attn","label":"Relative-position self-attention"},{"id":"r1","label":"Residual add"},{"id":"n2","label":"LayerNorm"},{"id":"ff","label":"Linear / SiLU / linear"},{"id":"r2","label":"Residual add"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"s3gen-estimator":{"title":"S3Gen flow estimator","scope":"An input, middle and output stack combines time-conditioned residual CNNs with Transformer blocks and a skip connection. Turbo additionally mixes the interval end-time embedding for MeanFlow.","blocks":[{"id":"cat","label":"Concatenate noise / token features / speaker / reference mel"},{"id":"time","label":"Time embedding / optional MeanFlow interval mixer"},{"id":"input","label":"Input residual CNN + attention stack","expand":"s3gen-flow-stage"},{"id":"mid","label":"Middle residual CNN + attention stacks","expand":"s3gen-flow-stage"},{"id":"join","label":"Concatenate input-stage skip"},{"id":"output","label":"Output residual CNN + attention stack","expand":"s3gen-flow-stage"},{"id":"head","label":"Convolutional velocity prediction"}],"edges":[["cat","input"],["time","input"],["input","mid"],["time","mid"],["mid","join"],["input","join"],["join","output"],["time","output"],["output","head"]]},"s3gen-flow-stage":{"title":"Flow residual / attention stage","scope":"The residual CNN uses causal convolutions, channel LayerNorm and Mish. Transformer feed-forward layers use GELU. Causal convolutions do not imply that all attention is causal.","blocks":[{"id":"cnn","label":"Conv / LayerNorm / Mish"},{"id":"time","label":"Add projected time embedding"},{"id":"cnn2","label":"Conv / LayerNorm / Mish + residual"},{"id":"attn","label":"Pre-LayerNorm self-attention + residual"},{"id":"ff","label":"Pre-LayerNorm GELU FFN + residual"}],"edges":[["cnn","time"],["time","cnn2"],["cnn2","attn"],["attn","ff"]],"residuals":[["cnn","cnn2"]]},"auk-conditioning":{"title":"Qwen2.5-Omni contextual conditioning","scope":"AuK uses the thinker's hidden states, not sampled text. Layer states are normalized and combined with learned softmax weights and a learned scale.","blocks":[{"id":"emb","label":"Text embeddings + inserted audio features"},{"id":"tr","label":"Causal Qwen2.5 decoder stack","expand":"qwen2-layer"},{"id":"mix","label":"LayerNorm + learned layer mixture"},{"id":"out","label":"Contextual conditioning sequence"}],"edges":[["emb","tr"],["tr","mix"],["mix","out"]]},"auk-audio":{"title":"Omni audio conditioning tower","scope":"Attention operates within audio windows. Encoded windows are joined, average-pooled and projected into the thinker's embedding width.","blocks":[{"id":"mel","label":"Log-mel frontend"},{"id":"cnn","label":"Two Conv1D / GELU layers + subsampling"},{"id":"pos","label":"Sinusoidal positions"},{"id":"tr","label":"Windowed Transformer stack","expand":"auk-audio-block"},{"id":"out","label":"Average pooling / LayerNorm / projection"}],"edges":[["mel","cnn"],["cnn","pos"],["pos","tr"],["tr","out"]]},"auk-audio-block":{"title":"Omni audio Transformer block","scope":"The audio tower uses pre-LayerNorm attention and a GELU feed-forward network, rather than the thinker's RMSNorm and SwiGLU.","blocks":[{"id":"n1","label":"LayerNorm"},{"id":"attn","label":"Windowed self-attention"},{"id":"r1","label":"Residual add"},{"id":"n2","label":"LayerNorm"},{"id":"ff","label":"Linear / GELU / linear"},{"id":"r2","label":"Residual add"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"auk-flow":{"title":"AuK flow backbone","scope":"Reference latents prefix the noisy target. Double-stream blocks use separate audio/text projections with joint attention; single-stream blocks process their concatenation. Only target positions produce velocity predictions.","blocks":[{"id":"emb","label":"Latent projection + convolutional position embedding"},{"id":"txt","label":"Context projection + RMSNorm"},{"id":"time","label":"Sinusoidal time embedding + MLP"},{"id":"dual","label":"Dual-stream joint-attention stack","expand":"auk-flow-block"},{"id":"join","label":"Concatenate text + audio states"},{"id":"single","label":"Single-stream DiT stack","expand":"auk-flow-block"},{"id":"out","label":"Target slice + adaptive norm + velocity head"}],"edges":[["emb","dual"],["txt","dual"],["time","dual"],["dual","join"],["join","single"],["time","single"],["single","out"]]},"auk-flow-block":{"title":"Time-conditioned flow Transformer block","scope":"Dual-stream blocks perform the normalization, projections and feed-forward separately per stream, but share attention over concatenated Q/K/V. Single-stream blocks use the same operations on one merged sequence.","blocks":[{"id":"n1","label":"Time-adaptive LayerNorm"},{"id":"attn","label":"Q/K RMSNorm + rotary joint attention"},{"id":"r1","label":"Time-gated residual"},{"id":"n2","label":"Time-adaptive LayerNorm"},{"id":"ff","label":"SwiGLU feed-forward"},{"id":"r2","label":"Time-gated residual"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"auk-vae-encoder":{"title":"Continuous waveform encoding","scope":"A convolutional posterior produces mean and log-variance for 64-dimensional latent frames. This is a variational latent representation, not vector-quantized token IDs.","blocks":[{"id":"in","label":"Waveform convolution + LeakyReLU"},{"id":"down","label":"Strided convolution stages"},{"id":"res","label":"Dilated residual CNN blocks"},{"id":"out","label":"Posterior projection + latent sampling"}],"edges":[["in","down"],["down","res"],["res","out"]]},"auk-vae-decoder":{"title":"BigVGAN-style latent waveform decoder","scope":"The decoder accepts continuous VAE latents rather than mel features. Each upsampling stage aggregates dilated residual branches with alias-free periodic activations.","blocks":[{"id":"in","label":"Latent input convolution"},{"id":"up","label":"Transposed convolution stages"},{"id":"res","label":"Alias-free SnakeBeta + dilated residual CNN"},{"id":"out","label":"Output convolution + waveform"}],"edges":[["in","up"],["up","res"],["res","out"]]},"qwen3-depth":{"title":"Qwen3 codebook-depth generation","scope":"The temporal talker supplies a hidden state and the first code. A separate causal Qwen3 decoder predicts the remaining codebooks for that frame; it is not another temporal speech generator.","blocks":[{"id":"seed","label":"Talker hidden state + first-code embedding"},{"id":"project","label":"Project to code-predictor width"},{"id":"stack","label":"Qwen3 causal decoder stack","expand":"qwen3-layer"},{"id":"head","label":"Codebook-specific output head + sampling"},{"id":"frame","label":"Complete frame / summed embedding feedback"}],"edges":[["seed","project"],["project","stack"],["stack","head"],["head","frame"]]},"qwen3-speaker":{"title":"ECAPA-TDNN speaker embedding","scope":"This branch summarizes reference identity into one embedding. It is separate from the speech tokenizer's framewise audio codes.","blocks":[{"id":"mel","label":"Log-mel frontend"},{"id":"tdnn","label":"Initial TDNN / ReLU"},{"id":"res","label":"SE-Res2Net stack","expand":"qwen3-speaker-block"},{"id":"mfa","label":"Concatenate multiscale features + TDNN"},{"id":"pool","label":"Attentive mean / standard-deviation pooling"},{"id":"proj","label":"Speaker embedding projection"}],"edges":[["mel","tdnn"],["tdnn","res"],["res","mfa"],["mfa","pool"],["pool","proj"]]},"qwen3-speaker-block":{"title":"SE-Res2Net speaker block","scope":"Channel groups use chained dilated temporal convolutions. Squeeze-excitation gates channels using global temporal context.","blocks":[{"id":"in","label":"Pointwise TDNN / ReLU"},{"id":"res2","label":"Split channels + chained dilated TDNNs"},{"id":"out","label":"Concatenate + pointwise TDNN / ReLU"},{"id":"se","label":"Temporal pooling + channel gating"},{"id":"add","label":"Residual add"}],"edges":[["in","res2"],["res2","out"],["out","se"],["se","add"]],"residuals":[["in","add"]]},"breeze-text":{"title":"T5Gemma2 text conditioning","scope":"Each text segment is encoded and projected into the temporal decoder width. The current C++ path uses bidirectional attention with layer-specific rotary frequencies.","blocks":[{"id":"embed","label":"Scaled text embedding"},{"id":"tr","label":"T5Gemma2 encoder stack","expand":"breeze-text-block"},{"id":"out","label":"Final RMSNorm + projection"}],"edges":[["embed","tr"],["tr","out"]]},"breeze-text-block":{"title":"T5Gemma2 encoder block","scope":"Gemma-style RMS normalization surrounds both sublayers. Q and K are normalized before rotary attention.","blocks":[{"id":"n1","label":"Pre-attention RMSNorm"},{"id":"attn","label":"Q/K norm + bidirectional rotary attention"},{"id":"p1","label":"Post-attention RMSNorm + residual"},{"id":"n2","label":"Pre-FFN RMSNorm"},{"id":"ff","label":"Gated GELU feed-forward"},{"id":"p2","label":"Post-FFN RMSNorm + residual"}],"edges":[["n1","attn"],["attn","p1"],["p1","n2"],["n2","ff"],["ff","p2"]],"residuals":[["n1","p1"],["n2","p2"]]},"breeze-depth":{"title":"Within-frame codebook generation","scope":"The cache resets for each audio frame. The temporal decoder predicts codebook zero; depth decoding fills the other fifteen. Summed codebook embeddings feed the next temporal step.","blocks":[{"id":"seed","label":"Project temporal state + first-code embedding"},{"id":"tr","label":"Causal Llama-style depth stack","expand":"llama-layer"},{"id":"head","label":"Position-specific codebook head + sampling"},{"id":"frame","label":"Assemble all 16 codebooks"}],"edges":[["seed","tr"],["tr","head"],["head","frame"]]},"breeze-reference":{"title":"Reference audio tokenization","scope":"Mimi-derived Qwen3-TTS tokenizer architecture. Semantic and acoustic branches have separate projections and codebooks. These framewise codes are distinct from a pooled speaker embedding.","blocks":[{"id":"cnn","label":"Causal SEANet residual CNN + downsampling"},{"id":"tr","label":"Rotary Transformer stack","expand":"mimi-transformer"},{"id":"rate","label":"Temporal downsampling"},{"id":"sem","label":"Semantic projection + one VQ codebook"},{"id":"ac","label":"Acoustic projection + residual VQ"},{"id":"codes","label":"Reference audio code frames"}],"edges":[["cnn","tr"],["tr","rate"],["rate","sem"],["rate","ac"],["sem","codes"],["ac","codes"]]},"breeze-codec":{"title":"Speech-code waveform decoding","scope":"Separate semantic and acoustic embeddings are summed after projection. The decoder reconstructs 24 kHz mono speech and clamps the output waveform.","blocks":[{"id":"lookup","label":"Codebook lookup + projection + sum"},{"id":"input","label":"Input convolution"},{"id":"tr","label":"Causal rotary Transformer","expand":"breeze-codec-transformer"},{"id":"up","label":"Transposed convolution + ConvNeXt stages","expand":"breeze-codec-convnext"},{"id":"wave","label":"SnakeBeta waveform upsampler","expand":"breeze-codec-wave"}],"edges":[["lookup","input"],["input","tr"],["tr","up"],["up","wave"]]},"breeze-codec-transformer":{"title":"Codec Transformer block","scope":"Sliding causal rotary attention and SwiGLU use RMSNorm and learned LayerScale residuals. Unlike the temporal Qwen3 generator, this codec block does not use Q/K normalization.","blocks":[{"id":"n1","label":"RMSNorm"},{"id":"attn","label":"Causal rotary grouped-query attention"},{"id":"r1","label":"LayerScale + residual"},{"id":"n2","label":"RMSNorm"},{"id":"ff","label":"SwiGLU feed-forward"},{"id":"r2","label":"LayerScale + residual"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"breeze-codec-convnext":{"title":"Codec ConvNeXt block","scope":"ConvNeXt refines features after each temporal upsampling stage.","blocks":[{"id":"dw","label":"Depthwise temporal convolution"},{"id":"norm","label":"LayerNorm"},{"id":"ff","label":"Pointwise expansion / GELU / projection"},{"id":"out","label":"LayerScale + residual"}],"edges":[["dw","norm"],["norm","ff"],["ff","out"]],"residuals":[["dw","out"]]},"breeze-codec-wave":{"title":"Waveform upsampling CNN","scope":"Each resolution stage uses SnakeBeta activation, transposed convolution and dilated residual units. Final convolution produces the waveform without an ISTFT.","blocks":[{"id":"in","label":"Causal input convolution"},{"id":"up","label":"SnakeBeta + transposed convolution"},{"id":"res","label":"SnakeBeta residual units: dilations 1 / 3 / 9"},{"id":"out","label":"SnakeBeta + output convolution + clamp"}],"edges":[["in","up"],["up","res"],["res","out"]]},"stable-t5":{"title":"T5 encoder conditioning","scope":"Foundation uses T5 encoder states only. Its base encoder has relative attention bias and a ReLU feed-forward; it does not use the rotary Gemma blocks shown for SA3.","blocks":[{"id":"tok","label":"SentencePiece + token embedding"},{"id":"stack","label":"Bidirectional T5 encoder stack","expand":"stable-t5-block"},{"id":"norm","label":"Final RMSNorm + padding mask"}],"edges":[["tok","stack"],["stack","norm"]]},"stable-t5-block":{"title":"T5-base encoder block","scope":"Pre-normalized bidirectional attention uses learned relative-position buckets, not rotary positions.","blocks":[{"id":"n1","label":"RMSNorm"},{"id":"attn","label":"Self-attention + relative-position bias"},{"id":"r1","label":"Residual add"},{"id":"n2","label":"RMSNorm"},{"id":"ff","label":"Linear / ReLU / linear"},{"id":"r2","label":"Residual add"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"stable-t5gemma":{"title":"T5Gemma encoder conditioning","scope":"SA3 consumes the encoder half of T5Gemma. It is bidirectional text encoding, not autoregressive text generation.","blocks":[{"id":"tok","label":"Checkpoint tokenizer + scaled embeddings"},{"id":"stack","label":"Gemma-style encoder stack","expand":"stable-t5gemma-block"},{"id":"norm","label":"Final RMSNorm + padding handling"}],"edges":[["tok","stack"],["stack","norm"]]},"stable-t5gemma-block":{"title":"T5Gemma encoder block","scope":"Four normalization sites surround attention and gated GELU feed-forward sublayers. Rotary attention is bidirectional with checkpoint-configured logit softcapping.","blocks":[{"id":"n1","label":"Pre-attention RMSNorm"},{"id":"attn","label":"Bidirectional rotary attention"},{"id":"post1","label":"Post-attention RMSNorm"},{"id":"r1","label":"Residual add"},{"id":"n2","label":"Pre-FFN RMSNorm"},{"id":"ff","label":"Gated GELU feed-forward"},{"id":"post2","label":"Post-FFN RMSNorm"},{"id":"r2","label":"Residual add"}],"edges":[["n1","attn"],["attn","post1"],["post1","r1"],["r1","n2"],["n2","ff"],["ff","post2"],["post2","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"stable-timing":{"title":"Timing embeddings","scope":"Foundation embeds start and total duration separately. SA3 uses total duration. This conditioning describes the requested sound length, separately from the diffusion solver timestep.","blocks":[{"id":"seconds","label":"Normalize timing values"},{"id":"fourier","label":"Fourier features"},{"id":"project","label":"Learned timing projection"},{"id":"cross","label":"Append timing tokens to text memory"},{"id":"global","label":"Global duration condition"}],"edges":[["seconds","fourier"],["fourier","project"],["project","cross"],["project","global"]]},"stable-foundation-flow":{"title":"Foundation diffusion backbone","scope":"A time/duration conditioning token is prepended to the noisy latent sequence. It is removed after the Transformer, before projecting the model prediction back into latent space.","blocks":[{"id":"proj","label":"Residual latent preprocessing + projection"},{"id":"time","label":"Diffusion time + global timing embedding"},{"id":"prepend","label":"Prepend condition token"},{"id":"tr","label":"Cross-attending Transformer stack","expand":"stable-foundation-block"},{"id":"out","label":"Remove prefix + latent output projection"}],"edges":[["proj","prepend"],["time","prepend"],["prepend","tr"],["tr","out"]]},"stable-foundation-block":{"title":"Foundation RF-DiT block","scope":"Noncausal rotary self-attention, text/timing cross-attention and SwiGLU each use pre-LayerNorm. This path does not use SA3's adaptive RMSNorm modulation.","blocks":[{"id":"n1","label":"LayerNorm"},{"id":"self","label":"Rotary self-attention"},{"id":"r1","label":"Residual add"},{"id":"n2","label":"LayerNorm"},{"id":"cross","label":"Text / timing cross-attention"},{"id":"r2","label":"Residual add"},{"id":"n3","label":"LayerNorm"},{"id":"ff","label":"SwiGLU feed-forward"},{"id":"r3","label":"Residual add"}],"edges":[["n1","self"],["self","r1"],["r1","n2"],["n2","cross"],["cross","r2"],["r2","n3"],["n3","ff"],["ff","r3"]],"residuals":[["n1","r1"],["n2","r2"],["n3","r3"]]},"stable3-flow":{"title":"Stable Audio 3 diffusion backbone","scope":"Noise or a noised source latent initializes generation. For inpainting, the keep mask and masked source latents additionally enter each block as local conditioning.","blocks":[{"id":"proj","label":"Residual latent preprocessing + projection"},{"id":"mem","label":"Prepend learned memory tokens"},{"id":"cond","label":"Time + duration modulation / text memory"},{"id":"tr","label":"Adaptive RF-DiT stack","expand":"stable3-block"},{"id":"out","label":"Remove memory tokens + latent prediction"}],"edges":[["proj","mem"],["mem","tr"],["cond","tr"],["tr","out"]]},"stable3-block":{"title":"Stable Audio 3 RF-DiT block","scope":"Self-attention and FFN residuals have sigmoid gates from time/duration modulation. Q/K are RMS-normalized. Differential attention is checkpoint-configured, not a separate generation stage.","blocks":[{"id":"n1","label":"Adaptive RMSNorm"},{"id":"self","label":"Rotary self-attention + Q/K norm"},{"id":"r1","label":"Gated residual add"},{"id":"cross","label":"RMSNorm + text cross-attention"},{"id":"r2","label":"Residual add + local masked-audio embedding"},{"id":"n3","label":"Adaptive RMSNorm"},{"id":"ff","label":"SwiGLU feed-forward"},{"id":"r3","label":"Gated residual add"}],"edges":[["n1","self"],["self","r1"],["r1","cross"],["cross","r2"],["r2","n3"],["n3","ff"],["ff","r3"]],"residuals":[["n1","r1"],["cross","r2"],["n3","r3"]]},"oobleck-encoder":{"title":"Oobleck waveform encoding","scope":"Source audio is compressed for initialization. The variational bottleneck supplies continuous latents rather than discrete codec IDs.","blocks":[{"id":"input","label":"Waveform input convolution"},{"id":"res","label":"Snake dilated residual units"},{"id":"down","label":"Strided convolution stages"},{"id":"out","label":"Latent distribution + sampling"}],"edges":[["input","res"],["res","down"],["down","out"]]},"oobleck-decoder":{"title":"Oobleck waveform decoding","scope":"Upsampling stages interleave transposed convolutions with dilated periodic-activation residual units.","blocks":[{"id":"input","label":"Latent input convolution"},{"id":"up","label":"Snake + transposed-convolution stages"},{"id":"res","label":"Snake dilated residual units"},{"id":"out","label":"Output convolution / stereo waveform"}],"edges":[["input","up"],["up","res"],["res","out"]]},"same-encoder":{"title":"SAME audio compression","scope":"Learned summary tokens pool local waveform patches. Small and Medium use checkpoint-specific local attention layouts; the output bottleneck scaling maps into diffusion coordinates.","blocks":[{"id":"patch","label":"Waveform patch mapping"},{"id":"summary","label":"Append learned summary tokens"},{"id":"tr","label":"Local rotary Transformer stack","expand":"same-block"},{"id":"select","label":"Select summary positions"},{"id":"latent","label":"Latent projection + bottleneck scaling"}],"edges":[["patch","summary"],["summary","tr"],["tr","select"],["select","latent"]]},"same-decoder":{"title":"SAME waveform reconstruction","scope":"Each latent is expanded into learned output-token positions. After local attention, the latent positions are discarded and output tokens are mapped into waveform patches. Checkpoint-configured bottleneck noise is separate from diffusion noise.","blocks":[{"id":"scale","label":"Inverse latent scaling / configured noise"},{"id":"proj","label":"Latent projection"},{"id":"expand","label":"Append learned output tokens per latent"},{"id":"tr","label":"Local rotary Transformer stack","expand":"same-block"},{"id":"select","label":"Retain output tokens"},{"id":"map","label":"Convolutional mapping + waveform unpatching"}],"edges":[["scale","proj"],["proj","expand"],["expand","tr"],["tr","select"],["select","map"]]},"same-block":{"title":"SAME Transformer block","scope":"Dynamic tanh normalization replaces LayerNorm/RMSNorm. Rotary attention can use differential attention. Decoder checkpoints may use sine-gated feed-forward layers near the output instead of SiLU gates.","blocks":[{"id":"n1","label":"Dynamic tanh normalization"},{"id":"attn","label":"Local rotary attention"},{"id":"r1","label":"Residual add"},{"id":"n2","label":"Dynamic tanh normalization"},{"id":"ff","label":"Gated SiLU / configured sine feed-forward"},{"id":"r2","label":"Residual add"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"mira-speaker":{"title":"Speaker context tokenization","scope":"Reference waveform preprocessing supplies fixed-duration mel features. Thirty-two learned Perceiver queries summarize ECAPA frame features; six scalar coordinates with four levels encode each context token.","blocks":[{"id":"mel","label":"Reference resampling / normalization / mel frontend"},{"id":"tdnn","label":"TDNN + SE-Res2 blocks","expand":"mira-ecapa-block"},{"id":"agg","label":"Concatenate multilevel features + projection"},{"id":"perc","label":"Learned-query Perceiver stack","expand":"mira-perceiver"},{"id":"norm","label":"RMSNorm + six-dimensional projection"},{"id":"fsq","label":"Finite scalar quantization: 32 context IDs"}],"edges":[["mel","tdnn"],["tdnn","agg"],["agg","perc"],["perc","norm"],["norm","fsq"]]},"mira-ecapa-block":{"title":"SE-Res2 TDNN block","scope":"Hierarchical channel groups and temporal dilations model reference speech. Squeeze/excitation uses mean-pooled channel context, not an autoregressive decoder.","blocks":[{"id":"tdnn","label":"Conv / ReLU / BatchNorm"},{"id":"res2","label":"Hierarchical Res2 split-channel TDNNs"},{"id":"merge","label":"Concatenate + TDNN"},{"id":"se","label":"Temporal mean / bottleneck / sigmoid gate"},{"id":"gate","label":"Channel gating"},{"id":"add","label":"Residual add"}],"edges":[["tdnn","res2"],["res2","merge"],["merge","se"],["merge","gate"],["se","gate"],["gate","add"]],"residuals":[["tdnn","add"]]},"mira-perceiver":{"title":"Speaker Perceiver layer","scope":"Learned latent queries attend to the concatenation of latent and reference-context states. The feed-forward uses gated GELU; quantization follows the complete stack.","blocks":[{"id":"attn","label":"Latent queries / latent + context keys and values"},{"id":"add","label":"Attention residual add"},{"id":"ff","label":"GEGLU feed-forward"},{"id":"out","label":"Feed-forward residual add"}],"edges":[["attn","add"],["add","ff"],["ff","out"]],"residuals":[["attn","add"],["ff","out"]]},"mira-processor":{"title":"Conditional acoustic reconstruction","scope":"Context IDs are dequantized and projected into a speaker vector. Despite upstream names containing Vocos, this processor outputs waveform-decoder features, not an ISTFT spectrum.","blocks":[{"id":"speech","label":"Speech-code lookup + projection"},{"id":"context","label":"Context-code lookup / flatten / speaker projection"},{"id":"plain","label":"Plain ConvNeXt stages","expand":"mira-convnext"},{"id":"conditional","label":"Speaker-conditioned ConvNeXt stack","expand":"mira-convnext-conditioned"},{"id":"out","label":"Final LayerNorm + projection + speaker addition"}],"edges":[["speech","plain"],["plain","conditional"],["context","conditional"],["conditional","out"],["context","out"]]},"mira-convnext":{"title":"Acoustic ConvNeXt block","scope":"A depthwise temporal convolution mixes local speech-token features. The channel MLP uses GELU and learned layer scaling.","blocks":[{"id":"dw","label":"Depthwise temporal convolution"},{"id":"norm","label":"LayerNorm"},{"id":"ff","label":"Linear / GELU / linear"},{"id":"scale","label":"Learned channel scale"},{"id":"add","label":"Residual add"}],"edges":[["dw","norm"],["norm","ff"],["ff","scale"],["scale","add"]],"residuals":[["dw","add"]]},"mira-convnext-conditioned":{"title":"Speaker-modulated ConvNeXt block","scope":"The reference speaker vector supplies LayerNorm scale and shift. This is deterministic conditional convolution, not a diffusion Transformer.","blocks":[{"id":"dw","label":"Depthwise temporal convolution"},{"id":"norm","label":"LayerNorm + speaker scale / shift"},{"id":"ff","label":"Linear / GELU / linear"},{"id":"scale","label":"Learned channel scale"},{"id":"add","label":"Residual add"}],"edges":[["dw","norm"],["norm","ff"],["ff","scale"],["scale","add"]],"residuals":[["dw","add"]]},"mira-decoder":{"title":"DAC-style waveform synthesis","scope":"Each upsampling stage contains residual units with dilations 1, 3 and 9. The output is 16 kHz; a separate FlashSR network raises it to 48 kHz.","blocks":[{"id":"in","label":"Input convolution"},{"id":"up","label":"Snake + transposed-convolution upsampling stages"},{"id":"res","label":"Snake / dilated convolution residual units"},{"id":"out","label":"Snake + output convolution + tanh"}],"edges":[["in","up"],["up","res"],["res","out"]]},"dots-patch":{"title":"Continuous patch feedback encoder","scope":"The generated flow patch is fed through this encoder before the next Qwen step. Reference/source patches prefill the same encoder. This recurrent feedback is represented here rather than as a cycle in the overview.","blocks":[{"id":"down","label":"Stride-2 causal latent convolution"},{"id":"proj","label":"Input projection"},{"id":"tr","label":"Causal rotary Transformer stack","expand":"dots-patch-block"},{"id":"pack","label":"Group patch positions"},{"id":"out","label":"Project to one Qwen embedding per patch"}],"edges":[["down","proj"],["proj","tr"],["tr","pack"],["pack","out"]]},"dots-patch-block":{"title":"Dots patch Transformer block","scope":"RMS-normalized causal attention uses rotary positions and checkpoint-configured Q/K normalization. The feed-forward is a two-layer SiLU MLP, not SwiGLU.","blocks":[{"id":"n1","label":"RMSNorm"},{"id":"attn","label":"Causal attention + RoPE"},{"id":"r1","label":"Residual add"},{"id":"n2","label":"RMSNorm"},{"id":"ff","label":"Linear / SiLU / linear"},{"id":"r2","label":"Residual add"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"dots-flow":{"title":"Autoregressive patch flow generation","scope":"Previous positions are causal; the current conditioning and noisy patch attend within the current block. SOAR uses iterative flow integration, while MeanFlow adds interval conditioning. Each completed patch feeds the semantic patch encoder for the next AR step.","blocks":[{"id":"prefix","label":"Projected Qwen state + prior acoustic patches"},{"id":"noise","label":"Noisy next-patch projection"},{"id":"condition","label":"Time + speaker + optional interval embedding"},{"id":"dit","label":"Block-masked rotary DiT stack","expand":"dots-flow-block"},{"id":"out","label":"Adaptive LayerNorm + velocity projection"},{"id":"solve","label":"Flow integration / next audio patch"}],"edges":[["prefix","dit"],["noise","dit"],["condition","dit"],["dit","out"],["out","solve"]]},"dots-flow-block":{"title":"Dots flow Transformer block","scope":"LayerNorm modulation supplies shift, scale and residual gates. Q/K use RMS normalization; the feed-forward uses GELU, unlike the Qwen backbone's SwiGLU.","blocks":[{"id":"n1","label":"Adaptive LayerNorm"},{"id":"attn","label":"Q/K RMSNorm + rotary block attention"},{"id":"r1","label":"Gated residual add"},{"id":"n2","label":"Adaptive LayerNorm"},{"id":"ff","label":"Linear / GELU / linear"},{"id":"r2","label":"Gated residual add"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"dots-vae-encoder":{"title":"AudioVAE continuous encoding","scope":"Reference/source audio is encoded only when acoustic context is needed. Speaker-only cloning uses CAM++ instead. The latent distribution is continuous; there is no codebook lookup.","blocks":[{"id":"cnn","label":"Causal convolution + LeakyReLU"},{"id":"down","label":"Strided downsampling + dilated residual stacks"},{"id":"lstm","label":"Projection / residual LSTM / projection"},{"id":"dist","label":"Latent distribution projection"},{"id":"sample","label":"Sample + normalize continuous latents"}],"edges":[["cnn","down"],["down","lstm"],["lstm","dist"],["dist","sample"]]},"dots-vae-decoder":{"title":"AudioVAE waveform synthesis","scope":"Continuous latents are decoded into 48 kHz mono audio. The decoder includes recurrent state before BigVGAN-style upsampling; it is not simply a mel vocoder. The checkpoint selects final tanh or clamping.","blocks":[{"id":"proj","label":"Latent projection"},{"id":"lstm","label":"Projection / residual LSTM / projection"},{"id":"conv","label":"Waveform decoder input convolution"},{"id":"up","label":"Transposed-convolution upsampling"},{"id":"res","label":"Alias-free SnakeBeta multi-receptive residual blocks"},{"id":"out","label":"Output activation + convolution + waveform bound"}],"edges":[["proj","lstm"],["lstm","conv"],["conv","up"],["up","res"],["res","out"]]},"heartmula-prompt":{"title":"HeartMuLa text prompt","scope":"The current audio.cpp route requires lyrics and tags and does not accept reference audio. The reserved continuous-conditioning vector is zero, not the output of an active music encoder.","blocks":[{"id":"tok","label":"Tokenize tags + lyrics"},{"id":"layout","label":"Boundary tokens + prompt masks"},{"id":"embed","label":"Text embeddings + projected zero conditioning"}],"edges":[["tok","layout"],["layout","embed"]]},"heartmula-temporal":{"title":"Temporal AR backbone","scope":"The temporal cache persists across frames. All codebook embeddings from the preceding frame are summed for the next temporal step; the first code and hidden state seed the depth decoder.","blocks":[{"id":"embed","label":"Masked text / summed audio embeddings"},{"id":"stack","label":"Llama-style causal Transformer stack","expand":"heartmula-llama-block"},{"id":"norm","label":"Final RMSNorm"},{"id":"head","label":"First-codebook head + sampling"},{"id":"state","label":"Last hidden state","kind":"output"}],"edges":[["embed","stack"],["stack","norm"],["norm","head"],["norm","state"]]},"heartmula-llama-block":{"title":"HeartMuLa Llama-style block","scope":"Pre-normalized causal grouped-query attention uses scaled rotary positions. There is no Q/K normalization. Temporal and depth models use this architecture with separate weights and sizes.","blocks":[{"id":"n1","label":"RMSNorm"},{"id":"attn","label":"Causal GQA + scaled RoPE"},{"id":"r1","label":"Residual add"},{"id":"n2","label":"RMSNorm"},{"id":"ff","label":"SwiGLU feed-forward"},{"id":"r2","label":"Residual add"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"heartmula-depth":{"title":"Within-frame codebook AR","scope":"The initial two positions are the temporal hidden state and first-code embedding. Later steps consume the preceding code embedding. The depth cache resets for every new audio frame.","blocks":[{"id":"seed","label":"Temporal state + first-code embedding"},{"id":"proj","label":"Depth input projection"},{"id":"stack","label":"Small Llama-style causal stack","expand":"heartmula-llama-block"},{"id":"head","label":"Position-specific codebook head"},{"id":"sample","label":"Sample remaining codebooks sequentially"}],"edges":[["seed","proj"],["proj","stack"],["stack","head"],["head","sample"]]},"heartcodec":{"title":"HeartCodec reconstruction","scope":"Flow matching reconstructs codec latents, not the song's symbolic structure. Stereo waveform chunks are assembled after scalar decoding.","blocks":[{"id":"vq","label":"Sum codebook lookups + output projection"},{"id":"cond","label":"Condition projection + temporal repetition"},{"id":"flow","label":"Conditional latent flow solve","expand":"heartcodec-flow"},{"id":"scalar","label":"Scalar quantization + waveform decoder","expand":"heartcodec-scalar"},{"id":"assemble","label":"Stereo chunk / overlap assembly"}],"edges":[["vq","cond"],["cond","flow"],["flow","scalar"],["scalar","assemble"]]},"heartcodec-flow":{"title":"HeartCodec two-stage flow estimator","scope":"Each flow step conditions on time, codec features and in-context latents. The second Transformer stage receives a concatenation with the original input representation and uses a wider hidden dimension.","blocks":[{"id":"input","label":"Noisy + context + codec-condition features"},{"id":"proj","label":"Input projection"},{"id":"stage1","label":"Modulated Transformer stage 1","expand":"heartcodec-flow-block"},{"id":"norm1","label":"Time-modulated final LayerNorm"},{"id":"concat","label":"Concatenate original hidden + stage 1"},{"id":"connect","label":"Wider connection projection"},{"id":"stage2","label":"Modulated Transformer stage 2","expand":"heartcodec-flow-block"},{"id":"out","label":"Time-modulated LayerNorm + velocity projection"}],"edges":[["input","proj"],["proj","stage1"],["stage1","norm1"],["norm1","concat"],["proj","concat"],["concat","connect"],["connect","stage2"],["stage2","out"]]},"heartcodec-flow-block":{"title":"HeartCodec flow Transformer block","scope":"Time produces shift, scale and residual gates. Attention is noncausal; this block differs from the causal AR generator even though both use rotary attention and gated feed-forward layers.","blocks":[{"id":"norm1","label":"Time-adaptive RMSNorm"},{"id":"attn","label":"Noncausal rotary self-attention"},{"id":"r1","label":"Time-gated residual add"},{"id":"norm2","label":"Time-adaptive RMSNorm"},{"id":"ff","label":"SwiGLU feed-forward"},{"id":"r2","label":"Time-gated residual add"}],"edges":[["norm1","attn"],["attn","r1"],["r1","norm2"],["norm2","ff"],["ff","r2"]],"residuals":[["norm1","r1"],["norm2","r2"]]},"heartcodec-scalar":{"title":"Scalar latent waveform decoder","scope":"Convolutional reconstruction after flow and scalar quantization. Upsampling stages contain PReLU dilated residual units; this is not a standalone Transformer codec decoder.","blocks":[{"id":"scalar","label":"Scalar latent quantization"},{"id":"input","label":"Input convolution"},{"id":"up","label":"Transposed-convolution upsampling stages"},{"id":"res","label":"PReLU + causal dilated residual convolutions"},{"id":"post","label":"Configured postprocessing + output convolution"}],"edges":[["scalar","input"],["input","up"],["up","res"],["res","post"]]},"personaplex-text":{"title":"Persona text prefix","scope":"Persona instructions condition the conversational model before user audio. This is not a separate text encoder or a target transcript for TTS.","blocks":[{"id":"wrap","label":"System/persona tags"},{"id":"tok","label":"SentencePiece tokenization"},{"id":"prefill","label":"Replay text tokens into temporal AR state"}],"edges":[["wrap","tok"],["tok","prefill"]]},"personaplex-preset":{"title":"Packaged voice prompt","scope":"Stored prompt embeddings bypass raw-reference Mimi encoding. Audio codebook delay state is restored alongside prompt replay.","blocks":[{"id":"id","label":"Voice ID selection"},{"id":"load","label":"Stored embeddings + delay state"},{"id":"replay","label":"Replay prompt into temporal AR state"}],"edges":[["id","load"],["load","replay"]]},"moshi-temporal":{"title":"Moshi temporal AR generation","scope":"User and assistant audio streams use delayed codebooks. The temporal state advances across frames, while the depth model generates one frame's codebooks. Generated codes feed subsequent temporal steps.","blocks":[{"id":"embed","label":"Sum text + user/assistant codebook embeddings"},{"id":"blocks","label":"Causal rotary Transformer stack","expand":"moshi-temporal-block"},{"id":"norm","label":"Final RMSNorm"},{"id":"head","label":"Text-token logits / sampling"},{"id":"hidden","label":"Hidden state for depth AR","kind":"output"}],"edges":[["embed","blocks"],["blocks","norm"],["norm","head"],["norm","hidden"]]},"moshi-temporal-block":{"title":"Moshi temporal block","scope":"Pre-RMSNorm causal attention uses RoPE, without Q/K normalization. The feed-forward is gated SiLU; this is Moshi, not a pretrained Qwen model.","blocks":[{"id":"n1","label":"RMSNorm"},{"id":"attn","label":"Causal RoPE attention + temporal KV state"},{"id":"r1","label":"Residual add"},{"id":"n2","label":"RMSNorm"},{"id":"ff","label":"SwiGLU feed-forward"},{"id":"r2","label":"Residual add"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"moshi-depth":{"title":"Within-frame depth AR","scope":"The first depth step embeds the selected text token; later steps embed the preceding audio code. Projections, attention/feed-forward weights and heads depend on codebook position. Depth state is separate from temporal conversation state.","blocks":[{"id":"lm","label":"Codebook-specific temporal-state projection"},{"id":"token","label":"Text / preceding codebook embedding"},{"id":"sum","label":"Add conditioning"},{"id":"stack","label":"Depth Transformer stack","expand":"moshi-depth-block"},{"id":"head","label":"Codebook-specific logits / sample / next depth"}],"edges":[["lm","sum"],["token","sum"],["sum","stack"],["stack","head"]]},"moshi-depth-block":{"title":"Moshi depth block","scope":"Causal attention spans codebook positions within the frame. There is no RoPE or Q/K normalization; codebook-specific projections encode the depth position.","blocks":[{"id":"n1","label":"RMSNorm"},{"id":"attn","label":"Causal depth attention"},{"id":"r1","label":"Residual add"},{"id":"n2","label":"RMSNorm"},{"id":"ff","label":"Codebook-specific SwiGLU"},{"id":"r2","label":"Residual add"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"mimi-encoder":{"title":"Mimi quantized audio encoding","scope":"PersonaPlex uses eight active audio codebooks. Semantic and acoustic projections are quantized separately; this is different from PocketTTS, which consumes continuous Mimi latents.","blocks":[{"id":"conv","label":"Causal ELU residual CNN + strided downsampling"},{"id":"attn","label":"Causal rotary Transformer","expand":"mimi-transformer"},{"id":"rate","label":"Rate-reduction convolution"},{"id":"sem","label":"Semantic projection + codebook quantization"},{"id":"acoustic","label":"Acoustic projection + residual quantization"},{"id":"ids","label":"Semantic + acoustic IDs","kind":"output"}],"edges":[["conv","attn"],["attn","rate"],["rate","sem"],["rate","acoustic"],["sem","ids"],["acoustic","ids"]]},"mimi-transformer":{"title":"Mimi streaming Transformer block","scope":"Pre-LayerNorm causal rotary attention and feed-forward residuals use learned LayerScale. Separate encoder and decoder weights retain temporal state across audio chunks.","blocks":[{"id":"n1","label":"LayerNorm"},{"id":"attn","label":"Causal RoPE attention + cached history"},{"id":"r1","label":"LayerScale + residual add"},{"id":"n2","label":"LayerNorm"},{"id":"ff","label":"Feed-forward network"},{"id":"r2","label":"LayerScale + residual add"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"mimi-decoder":{"title":"Mimi code-to-waveform decoding","scope":"Semantic and acoustic codebook vectors are projected and added. Streaming keeps attention history, convolution history and transposed-convolution overlap.","blocks":[{"id":"codes","label":"Semantic / acoustic codebook lookup + sum"},{"id":"proj","label":"Branch projections + latent addition"},{"id":"rate","label":"Depthwise rate upsampling"},{"id":"attn","label":"Causal rotary Transformer","expand":"mimi-transformer"},{"id":"conv","label":"ELU residual CNN / transposed-convolution stages"},{"id":"out","label":"Output waveform convolution"}],"edges":[["codes","proj"],["proj","rate"],["rate","attn"],["attn","conv"],["conv","out"]]},"controlfoley-clip-text":{"title":"OpenCLIP text conditioning","scope":"The integration retains text-token features rather than sampling text. The block uses pre-LayerNorm attention and QuickGELU feed-forward residuals.","blocks":[{"id":"tok","label":"Byte BPE + boundary / padding tokens"},{"id":"embed","label":"Token + position embeddings"},{"id":"stack","label":"Causal Transformer blocks","expand":"controlfoley-clip-block"},{"id":"norm","label":"Final LayerNorm / token features"}],"edges":[["tok","embed"],["embed","stack"],["stack","norm"]]},"controlfoley-clip-vision":{"title":"OpenCLIP image conditioning","scope":"Each sampled frame is encoded by a noncausal vision Transformer. Its class-token projection supplies one visual embedding per sampled frame.","blocks":[{"id":"patch","label":"Image patch convolution"},{"id":"pos","label":"Class token + positions / pre-norm"},{"id":"stack","label":"Vision Transformer blocks","expand":"controlfoley-clip-block"},{"id":"pool","label":"Final norm + class-token selection"},{"id":"proj","label":"Projection to visual embedding"}],"edges":[["patch","pos"],["pos","stack"],["stack","pool"],["pool","proj"]]},"controlfoley-clip-block":{"title":"OpenCLIP residual block","scope":"Text uses a causal attention mask; image patches attend bidirectionally. Weights are separate even though the block structure is shared.","blocks":[{"id":"n1","label":"LayerNorm"},{"id":"attn","label":"Multi-head self-attention"},{"id":"r1","label":"Residual add"},{"id":"n2","label":"LayerNorm"},{"id":"ff","label":"Linear + QuickGELU + linear"},{"id":"r2","label":"Residual add"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"controlfoley-cavmae":{"title":"CAV-MAE-ST visual branch","scope":"Only the visual path runs here. Visual-specific blocks precede shared pretrained blocks; both use pre-norm attention and GELU feed-forward residuals.","blocks":[{"id":"patch","label":"2D patch convolution"},{"id":"embed","label":"Position + visual modality embeddings"},{"id":"visual","label":"Visual Transformer blocks","expand":"controlfoley-encoder-block"},{"id":"shared","label":"Shared Transformer blocks","expand":"controlfoley-encoder-block"},{"id":"norm","label":"Final visual LayerNorm"}],"edges":[["patch","embed"],["embed","visual"],["visual","shared"],["shared","norm"]]},"controlfoley-encoder-block":{"title":"Pre-norm encoder block","scope":"Bidirectional self-attention with an exact-GELU feed-forward network. This structural diagram does not imply shared weights between encoders.","blocks":[{"id":"n1","label":"LayerNorm"},{"id":"attn","label":"Bidirectional self-attention"},{"id":"r1","label":"Residual add"},{"id":"n2","label":"LayerNorm"},{"id":"ff","label":"Linear + GELU + linear"},{"id":"r2","label":"Residual add"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"controlfoley-synchformer":{"title":"Synchformer video timing","scope":"Overlapping video segments are encoded using separate temporal and spatial attention. A spatial aggregator preserves temporal tokens for synchronization conditioning.","blocks":[{"id":"segment","label":"Overlapping frame segments"},{"id":"patch","label":"3D tubelet convolution + class / position tokens"},{"id":"time","label":"LayerNorm + temporal attention / residual"},{"id":"space","label":"LayerNorm + spatial attention / residual"},{"id":"ff","label":"LayerNorm + GELU FFN / residual / repeat"},{"id":"aggregate","label":"Spatial attention aggregator + norm"},{"id":"out","label":"Time-aligned video features","kind":"output"}],"edges":[["segment","patch"],["patch","time"],["time","space"],["space","ff"],["ff","aggregate"],["aggregate","out"]]},"controlfoley-clap":{"title":"CLAP reference-audio encoder","scope":"The audio branch uses hierarchical shifted-window attention, not a speech recognition decoder. Relative position biases and shifted-window masks are applied inside attention.","blocks":[{"id":"mel","label":"Resample + log-mel image"},{"id":"patch","label":"2D patch embedding + LayerNorm"},{"id":"swin","label":"Window / shifted-window Transformer blocks","expand":"controlfoley-swin-block"},{"id":"merge","label":"Patch merging / repeat stages"},{"id":"pool","label":"Final norm + mean pooling"},{"id":"proj","label":"Audio embedding projection"}],"edges":[["mel","patch"],["patch","swin"],["swin","merge"],["merge","pool"],["pool","proj"]]},"controlfoley-swin-block":{"title":"Audio Swin block","scope":"Alternating blocks shift the window partition. Patch merging between stages reduces spatial resolution and expands channels.","blocks":[{"id":"norm","label":"LayerNorm + optional window shift"},{"id":"partition","label":"Partition windows"},{"id":"attn","label":"Attention + relative bias + window mask"},{"id":"reverse","label":"Reverse windows / shift"},{"id":"r1","label":"Residual add"},{"id":"ff","label":"LayerNorm + GELU feed-forward"},{"id":"r2","label":"Residual add"}],"edges":[["norm","partition"],["partition","attn"],["attn","reverse"],["reverse","r1"],["r1","ff"],["ff","r2"]],"residuals":[["norm","r1"],["ff","r2"]]},"controlfoley-style":{"title":"Reference timbre conditioning","scope":"This is raw-waveform MERT from the Wav2Vec2/HuBERT family, not the mel-Conformer MERT2 used by SheetSage2. The final temporal mean removes timing detail from the style condition.","blocks":[{"id":"mert","label":"Raw-waveform MERT encoder","expand":"controlfoley-mert"},{"id":"proj","label":"Projection + sinusoidal positions"},{"id":"style","label":"Style Transformer","expand":"controlfoley-encoder-block"},{"id":"norm","label":"BatchNorm"},{"id":"vq","label":"Codebook quantization + temporal reduction"},{"id":"out","label":"Output projection + temporal mean"}],"edges":[["mert","proj"],["proj","style"],["style","norm"],["norm","vq"],["vq","out"]]},"controlfoley-mert":{"title":"MERT waveform encoder","scope":"The packaged style conditioner uses a 12-layer post-norm waveform Transformer. Strided convolution extracts acoustic frames; no speech transcription head or MERT2 Conformer is involved.","blocks":[{"id":"cnn","label":"Strided waveform CNN / first-layer GroupNorm / GELU"},{"id":"proj","label":"Feature LayerNorm + projection"},{"id":"pos","label":"Grouped positional convolution + input LayerNorm"},{"id":"stack","label":"Post-norm Transformer stack","expand":"controlfoley-mert-block"}],"edges":[["cnn","proj"],["proj","pos"],["pos","stack"]]},"controlfoley-mert-block":{"title":"MERT post-norm Transformer","scope":"The waveform MERT checkpoint uses bidirectional attention and GELU feed-forward layers with post-residual LayerNorm.","blocks":[{"id":"attn","label":"Bidirectional self-attention"},{"id":"r1","label":"Residual add + LayerNorm"},{"id":"ff","label":"Linear + GELU + linear"},{"id":"r2","label":"Residual add + LayerNorm"}],"edges":[["attn","r1"],["r1","ff"],["ff","r2"]],"residuals":[["attn","r1"],["ff","r2"]]},"controlfoley-flow":{"title":"Multimodal conditional flow","scope":"Absent modalities use learned empty conditioning. Synchronization tokens provide time-varying modulation; semantic and timbre features also feed a global condition. Euler integration generates continuous latents.","blocks":[{"id":"proj","label":"Project noisy latent / visual / text / audio streams"},{"id":"cond","label":"Timestep + global + sync modulation"},{"id":"joint","label":"Joint multimodal attention blocks","expand":"controlfoley-flow-block"},{"id":"single","label":"Latent-only attention blocks","expand":"controlfoley-flow-block"},{"id":"head","label":"Modulated norm + flow projection"},{"id":"solve","label":"Guided Euler steps / latent output"}],"edges":[["proj","joint"],["cond","joint"],["joint","single"],["cond","single"],["single","head"],["head","solve"]]},"controlfoley-flow-block":{"title":"Modulated flow attention block","scope":"Joint blocks concatenate Q/K/V across modality streams, then split attention outputs back into their streams. Later blocks operate only on latents. Rotary positions apply to latent and video streams.","blocks":[{"id":"norm","label":"LayerNorm + condition shift / scale"},{"id":"qkv","label":"Q/K/V + Q/K RMSNorm + optional RoPE"},{"id":"attn","label":"Joint or latent-only attention"},{"id":"r1","label":"Projection + condition-gated residual"},{"id":"norm2","label":"LayerNorm + condition shift / scale"},{"id":"ff","label":"Gated SiLU feed-forward"},{"id":"r2","label":"Condition-gated residual add"}],"edges":[["norm","qkv"],["qkv","attn"],["attn","r1"],["r1","norm2"],["norm2","ff"],["ff","r2"]],"residuals":[["norm","r1"],["norm2","r2"]]},"controlfoley-vae":{"title":"Mel-latent VAE decoder","scope":"Magnitude-preserving residual operations and SiLU convolutions reconstruct mel features. The output is denormalized with stored statistics before a separate BigVGAN vocoder.","blocks":[{"id":"in","label":"Latent input convolution"},{"id":"mid","label":"Residual CNN / attention / residual CNN"},{"id":"levels","label":"Residual convolution stages"},{"id":"up","label":"Nearest temporal upsample + convolution"},{"id":"out","label":"Scaled SiLU + output convolution"},{"id":"stats","label":"Mel mean / standard deviation restoration"}],"edges":[["in","mid"],["mid","levels"],["levels","up"],["up","out"],["out","stats"]]},"sheetsage-mel":{"title":"Music mel frontend","scope":"The checkpoint stores its filterbank and per-bin statistics. Overlapping audio windows are processed independently before event assembly.","blocks":[{"id":"mix","label":"Resample + mono channel mix"},{"id":"win","label":"Window audio"},{"id":"stft","label":"STFT + mel projection"},{"id":"log","label":"Log-power features"},{"id":"norm","label":"Per-bin mean / standard deviation"}],"edges":[["mix","win"],["win","stft"],["stft","log"],["log","norm"]]},"mert2":{"title":"MERT2 music encoder","scope":"This MERT2 path uses a mel frontend and Conformer blocks, not the raw-waveform HuBERT frontend of earlier MERT encoders. Learned layer weights mix the subsampling output and Conformer states.","blocks":[{"id":"sub","label":"ConvNeXt-style subsampling","expand":"mert2-subsampling"},{"id":"stack","label":"Rotary Conformer stack","expand":"mert2-conformer"},{"id":"mix","label":"Weighted mixture of intermediate states"},{"id":"mem","label":"Score-decoder memory","kind":"output"}],"edges":[["sub","stack"],["sub","mix"],["stack","mix"],["mix","mem"]]},"mert2-subsampling":{"title":"ConvNeXt-style temporal subsampling","scope":"Stages optionally normalize and stride-convolve between widths. Each residual block includes global response normalization after its expanded GELU projection.","blocks":[{"id":"sample","label":"Optional LayerNorm + stride convolution"},{"id":"dw","label":"Depthwise temporal convolution"},{"id":"norm","label":"LayerNorm"},{"id":"up","label":"Linear expansion + GELU"},{"id":"grn","label":"Global response normalization"},{"id":"down","label":"Linear projection"},{"id":"add","label":"Residual add / repeat blocks"}],"edges":[["sample","dw"],["dw","norm"],["norm","up"],["up","grn"],["grn","down"],["down","add"]],"residuals":[["dw","add"]]},"mert2-conformer":{"title":"MERT2 Conformer block","scope":"Two half-scaled GELU feed-forward residuals surround rotary self-attention and a convolution module. The convolution uses GLU, depthwise filtering, LayerNorm and GELU.","blocks":[{"id":"ff1","label":"LayerNorm + GELU FFN"},{"id":"r1","label":"Half-scale residual add"},{"id":"attn","label":"LayerNorm + rotary self-attention"},{"id":"r2","label":"Residual add"},{"id":"conv","label":"Norm + pointwise GLU / depthwise convolution"},{"id":"proj","label":"LayerNorm + GELU + pointwise projection"},{"id":"r3","label":"Residual add"},{"id":"ff2","label":"LayerNorm + GELU FFN"},{"id":"r4","label":"Half-scale residual add + final LayerNorm"}],"edges":[["ff1","r1"],["r1","attn"],["attn","r2"],["r2","conv"],["conv","proj"],["proj","r3"],["r3","ff2"],["ff2","r4"]],"residuals":[["ff1","r1"],["attn","r2"],["conv","r3"],["ff2","r4"]]},"sheetsage-decoder":{"title":"Score AR decoder","scope":"Audio memory is read by cross-attention. Unlike a Qwen-style pre-norm language block, this decoder normalizes after each residual sublayer.","blocks":[{"id":"embed","label":"Token + learned position embeddings / norm"},{"id":"blocks","label":"Cross-attention AR stack","expand":"sheetsage-decoder-block"},{"id":"head","label":"Symbol vocabulary projection"},{"id":"token","label":"Next symbolic token / repeat"}],"edges":[["embed","blocks"],["blocks","head"],["head","token"]]},"sheetsage-decoder-block":{"title":"Post-norm score decoder block","scope":"Self-attention is causal; cross-attention reads the full encoded audio window. Both and the GELU feed-forward use residual addition followed by LayerNorm.","blocks":[{"id":"self","label":"Causal self-attention"},{"id":"r1","label":"Residual add + LayerNorm"},{"id":"cross","label":"Audio cross-attention"},{"id":"r2","label":"Residual add + LayerNorm"},{"id":"ff","label":"Linear + GELU + linear"},{"id":"r3","label":"Residual add + LayerNorm"}],"edges":[["self","r1"],["r1","cross"],["cross","r2"],["r2","ff"],["ff","r3"]],"residuals":[["self","r1"],["cross","r2"],["ff","r3"]]},"sheetsage-events":{"title":"Score event assembly","scope":"Formatting turns model predictions into notation; it does not synthesize audio. Window timing and musical labels belong to this postprocessing stage.","blocks":[{"id":"decode","label":"Decode symbolic token vocabulary"},{"id":"events","label":"Assemble timed musical events"},{"id":"join","label":"Combine window predictions"},{"id":"abc","label":"Format readable ABC score"}],"edges":[["decode","events"],["events","join"],["join","abc"]]},"muscriptor-prefix":{"title":"MuScriptor conditioning prefix","scope":"The frontend is a linear projection, not a separate Transformer encoder. Instrument selection is optional; dataset conditioning uses a learned embedding.","blocks":[{"id":"mel","label":"Log-mel frames","kind":"input"},{"id":"proj","label":"Linear projection + validity mask"},{"id":"tags","label":"Dataset / instrument embeddings"},{"id":"cat","label":"Concatenate prefix tokens"}],"edges":[["mel","proj"],["proj","cat"],["tags","cat"]]},"muscriptor-ar":{"title":"MuScriptor AR event generation","scope":"The conditioning prefix and event-token embeddings share a decoder-only Transformer. Open notes from the previous chunk can be supplied as a prelude; no cross-attention encoder is needed.","blocks":[{"id":"cat","label":"Audio prefix + event/prelude embeddings"},{"id":"pos","label":"Sinusoidal position addition"},{"id":"blocks","label":"Pre-norm causal Transformer stack","expand":"muscriptor-ar-block"},{"id":"norm","label":"Final LayerNorm"},{"id":"head","label":"Event logits + selection / repeat"}],"edges":[["cat","pos"],["pos","blocks"],["blocks","norm"],["norm","head"]]},"muscriptor-ar-block":{"title":"MuScriptor Transformer block","scope":"Ordinary multi-head causal self-attention and an exact-GELU feed-forward network. Positions are added outside the block, rather than applied as rotary embeddings.","blocks":[{"id":"n1","label":"LayerNorm"},{"id":"attn","label":"Causal multi-head self-attention"},{"id":"r1","label":"Residual add"},{"id":"n2","label":"LayerNorm"},{"id":"ff","label":"Linear + GELU + linear"},{"id":"r2","label":"Residual add"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ff"],["ff","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"muscriptor-events":{"title":"Note-event decoding","scope":"The token representation provides timing, pitch and instrument, not expressive velocity prediction. Serialization is deterministic postprocessing, not a learned head.","blocks":[{"id":"tok","label":"Decode event vocabulary"},{"id":"state","label":"Track note starts / ends and instrument"},{"id":"join","label":"Apply chunk offsets / open-note continuity"},{"id":"write","label":"Serialize MIDI or note-event JSON"}],"edges":[["tok","state"],["state","join"],["join","write"]]},"dramabox-gemma":{"title":"Gemma3 prompt conditioning","scope":"The causal language backbone is used only to extract features. Normalized and projected hidden states are summed into audio conditioning; there is no audio-token sampling head here.","blocks":[{"id":"token","label":"Gemma tokenizer + scaled token embeddings"},{"id":"stack","label":"Gemma3 causal Transformer stack","expand":"gemma3-block"},{"id":"collect","label":"Collect layer hidden states"},{"id":"norm","label":"Per-state RMS normalization + padding mask"},{"id":"proj","label":"Layer projections + sum + bias"}],"edges":[["token","stack"],["stack","collect"],["collect","norm"],["norm","proj"]]},"gemma3-block":{"title":"Gemma3 decoder block","scope":"Local sliding-window and global causal attention layers use separate rotary settings. Q/K normalization and gated GELU distinguish this block from a generic Qwen/SwiGLU block.","blocks":[{"id":"n1","label":"Gemma RMSNorm"},{"id":"attn","label":"Q/K norm + RoPE + causal GQA"},{"id":"post1","label":"Post-attention RMSNorm"},{"id":"r1","label":"Residual add"},{"id":"n2","label":"Pre-FFN RMSNorm"},{"id":"ff","label":"Gated GELU feed-forward"},{"id":"post2","label":"Post-FFN RMSNorm"},{"id":"r2","label":"Residual add"}],"edges":[["n1","attn"],["attn","post1"],["post1","r1"],["r1","n2"],["n2","ff"],["ff","post2"],["post2","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"dramabox-connector":{"title":"Audio prompt connector","scope":"Valid prompt features are packed first; learned registers fill the remaining positions. Repeated noncausal blocks apply per-head gates as well as ordinary residual connections.","blocks":[{"id":"pack","label":"Prompt features + learned register fill"},{"id":"n1","label":"RMSNorm"},{"id":"attn","label":"Q/K RMSNorm + split RoPE attention"},{"id":"gate","label":"Per-head sigmoid gates + output projection"},{"id":"r1","label":"Residual add"},{"id":"n2","label":"RMSNorm"},{"id":"ff","label":"Linear + GELU + linear"},{"id":"r2","label":"Residual add / repeat blocks"},{"id":"norm","label":"Final RMSNorm"}],"edges":[["pack","n1"],["n1","attn"],["attn","gate"],["gate","r1"],["r1","n2"],["n2","ff"],["ff","r2"],["r2","norm"]],"residuals":[["n1","r1"],["n2","r2"]]},"dramabox-dit":{"title":"LTX audio flow Transformer","scope":"Noise latents evolve under flow matching. Reference latents, when supplied, remain fixed conditioning. Timestep modulation gates the residuals; attention also has learned head gates.","blocks":[{"id":"embed","label":"Noisy audio + optional reference latent projection"},{"id":"time","label":"Timestep embeddings / modulation"},{"id":"block","label":"Audio DiT stack","expand":"dramabox-dit-block"},{"id":"out","label":"Modulated final norm + velocity projection"},{"id":"solve","label":"Flow integration / repeat steps"}],"edges":[["embed","block"],["time","block"],["block","out"],["out","solve"]]},"dramabox-dit-block":{"title":"Audio DiT block","scope":"Text memory is projected by the connector. Each attention output includes a learned per-head gate before its timestep-gated residual; the feed-forward activation is GELU.","blocks":[{"id":"n1","label":"Timestep-adaptive RMSNorm"},{"id":"self","label":"Rotary self-attention + head gates"},{"id":"r1","label":"Time-gated residual add"},{"id":"n2","label":"Adaptive query / text normalization"},{"id":"cross","label":"Text cross-attention + head gates"},{"id":"r2","label":"Time-gated residual add"},{"id":"n3","label":"Timestep-adaptive RMSNorm"},{"id":"ff","label":"GELU feed-forward"},{"id":"r3","label":"Time-gated residual add"}],"edges":[["n1","self"],["self","r1"],["r1","n2"],["n2","cross"],["cross","r2"],["r2","n3"],["n3","ff"],["ff","r3"]],"residuals":[["n1","r1"],["n2","r2"],["n3","r3"]]},"dramabox-vae-encoder":{"title":"Mel AudioVAE encoder","scope":"Convolutions operate over time and mel frequency with causal time padding. The mean half of the posterior projection is used for reference conditioning, without sampling or vector quantization.","blocks":[{"id":"conv","label":"Causal 2D input convolution"},{"id":"down","label":"PixelNorm residual CNN + downsample stages"},{"id":"mid","label":"Two middle residual CNN blocks"},{"id":"norm","label":"PixelNorm + SiLU"},{"id":"proj","label":"Causal posterior convolution"},{"id":"mean","label":"Select mean + crop latent extent"}],"edges":[["conv","down"],["down","mid"],["mid","norm"],["norm","proj"],["proj","mean"]]},"dramabox-vae-decoder":{"title":"Mel AudioVAE decoder","scope":"This decoder returns stereo mel features, not waveform samples. It is a convolutional VAE with PixelNorm residual blocks rather than a token codec.","blocks":[{"id":"conv","label":"Causal 2D latent convolution"},{"id":"mid","label":"Two middle residual CNN blocks"},{"id":"up","label":"Residual CNN + causal upsample stages"},{"id":"norm","label":"PixelNorm + SiLU"},{"id":"out","label":"Causal 2D output convolution + crop"},{"id":"mel","label":"Stereo mel features","kind":"output"}],"edges":[["conv","mid"],["mid","up"],["up","norm"],["norm","out"],["out","mel"]]},"dramabox-bwe":{"title":"Residual bandwidth extension","scope":"Distinct from the first waveform vocoder. A second BigVGAN synthesizes the correction added to a filtered resampling of the low-rate waveform.","blocks":[{"id":"input","label":"Low-rate waveform","kind":"input"},{"id":"mel","label":"Stereo STFT / log-mel"},{"id":"gen","label":"BigVGAN residual generator","expand":"bigvgan"},{"id":"skip","label":"Filtered waveform resampling"},{"id":"sum","label":"Sum residual + skip / clamp"},{"id":"out","label":"High-rate waveform","kind":"output"}],"edges":[["input","mel"],["mel","gen"],["input","skip"],["gen","sum"],["skip","sum"],["sum","out"]]},"magpie-text":{"title":"Magpie multilingual text processing","scope":"Frontend choice is language-specific. The current audio.cpp integration has IPA/pinyin, character and byte paths; it does not expose the upstream Japanese frontend.","blocks":[{"id":"lang","label":"Language-specific normalization"},{"id":"tokens","label":"Phonemes / characters / byte IDs"},{"id":"vocab","label":"Checkpoint vocabulary offsets + boundary tokens"}],"edges":[["lang","tokens"],["tokens","vocab"]]},"magpie-encoder":{"title":"Magpie text encoder","scope":"The text Transformer is causal, not bidirectional. It uses learned absolute position embeddings and convolutional feed-forward layers, not RoPE/SwiGLU.","blocks":[{"id":"embed","label":"Text token + learned position embeddings"},{"id":"stack","label":"Causal Transformer stack","expand":"magpie-self-block"},{"id":"norm","label":"Final LayerNorm"},{"id":"kv","label":"Text cross-attention memory"}],"edges":[["embed","stack"],["stack","norm"],["norm","kv"]]},"magpie-self-block":{"title":"Magpie self-attention block","scope":"Shared structural pattern for text and local decoders, with different weights and feed-forward kernel widths. Text uses causal convolutional feed-forward layers; local codebook prediction uses pointwise layers.","blocks":[{"id":"n1","label":"LayerNorm"},{"id":"attn","label":"Causal self-attention"},{"id":"r1","label":"Residual addition"},{"id":"n2","label":"LayerNorm"},{"id":"ffn","label":"Convolution + GELU + convolution"},{"id":"r2","label":"Residual addition"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ffn"],["ffn","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"magpie-temporal":{"title":"Magpie temporal decoder","scope":"Speaker context precedes generated audio-stack embeddings. The temporal decoder attends to text memory and provides the hidden state for local codebook generation. Its text-attention alignment also guides stopping.","blocks":[{"id":"input","label":"Speaker context + previous frame-stack embeddings"},{"id":"pos","label":"Learned positions"},{"id":"stack","label":"Causal decoder with text cross-attention","expand":"magpie-cross-block"},{"id":"norm","label":"Final LayerNorm"},{"id":"hidden","label":"Local decoder conditioning"},{"id":"head","label":"Auxiliary audio-token / end logits"}],"edges":[["input","pos"],["pos","stack"],["stack","norm"],["norm","hidden"],["norm","head"]]},"magpie-cross-block":{"title":"Magpie cross-attention decoder block","scope":"Text keys/values can be prepared once. Alignment priors bias cross-attention without changing the input into a decoder-only language model.","blocks":[{"id":"n1","label":"LayerNorm"},{"id":"self","label":"Causal self-attention"},{"id":"r1","label":"Residual addition"},{"id":"n2","label":"Normalize query / text memory"},{"id":"cross","label":"Text cross-attention + alignment prior"},{"id":"r2","label":"Residual addition"},{"id":"n3","label":"LayerNorm"},{"id":"ffn","label":"Convolutional GELU feed-forward"},{"id":"r3","label":"Residual addition"}],"edges":[["n1","self"],["self","r1"],["r1","n2"],["n2","cross"],["cross","r2"],["r2","n3"],["n3","ffn"],["ffn","r3"]],"residuals":[["n1","r1"],["n2","r2"],["n3","r3"]]},"magpie-local":{"title":"Magpie local codebook generation","scope":"Autoregression runs across codebooks in each frame stack, separately from the temporal decoder's autoregression across audio stacks.","blocks":[{"id":"seed","label":"Temporal hidden state / preceding code embedding"},{"id":"pos","label":"Local learned position embedding"},{"id":"tr","label":"Local causal Transformer","expand":"magpie-self-block"},{"id":"head","label":"Position-specific codebook head"},{"id":"sample","label":"Sample stacked-frame codec IDs"}],"edges":[["seed","pos"],["pos","tr"],["tr","head"],["head","sample"]]},"magpie-preset":{"title":"Baked speaker conditioning","scope":"Voice names resolve to indices into packaged speaker context embeddings. No reference waveform or reference transcript is consumed by this route.","blocks":[{"id":"id","label":"Voice name / numeric speaker ID"},{"id":"lookup","label":"Speaker context embedding lookup"},{"id":"prefix","label":"Temporal decoder prefix"}],"edges":[["id","lookup"],["lookup","prefix"]]},"nemo-nanocodec":{"title":"NanoCodec FSQ waveform decoder","scope":"Code indices unpack into normalized finite scalar levels. The decoder uses Snake on part of its channels and LeakyReLU on the rest, plus grouped transposed convolutions and multi-branch residual convolutions.","blocks":[{"id":"fsq","label":"Mixed-radix FSQ index decoding"},{"id":"conv","label":"Causal input convolution"},{"id":"up","label":"Split Snake / LeakyReLU + grouped upsampling"},{"id":"res","label":"Multi-branch dilated residual convolutions"},{"id":"repeat","label":"Repeat upsampling stages"},{"id":"wave","label":"Split activation + waveform convolution / clamp"}],"edges":[["fsq","conv"],["conv","up"],["up","res"],["res","repeat"],["repeat","wave"]]},"sopro-text":{"title":"Sopro text frontend","scope":"Text and optional language information are converted into the semantic LM's text token sequence. This is distinct from the reference audio's semantic token vocabulary.","blocks":[{"id":"norm","label":"Minimal normalization / language marking"},{"id":"sp","label":"SentencePiece tokenizer"},{"id":"emb","label":"Text embedding table"}],"edges":[["norm","sp"],["sp","emb"]]},"sopro-semantic":{"title":"Sopro reference semantic encoder","scope":"Whisper-style mel Transformer architecture, not an ASR transcription route. A learned head selects per-dimension scalar levels and packs them into semantic IDs.","blocks":[{"id":"mel","label":"Semantic log-mel frontend"},{"id":"conv","label":"Conv1D + GELU subsampling"},{"id":"pos","label":"Add positional embeddings"},{"id":"tr","label":"Pre-LayerNorm Transformer stack","expand":"sopro-semantic-block"},{"id":"head","label":"Normalize + scalar-level prediction heads"},{"id":"ids","label":"Mixed-radix semantic token IDs"}],"edges":[["mel","conv"],["conv","pos"],["pos","tr"],["tr","head"],["head","ids"]]},"sopro-semantic-block":{"title":"Sopro semantic encoder block","scope":"Bidirectional attention with ordinary GELU feed-forward layers. This differs from the rotary SwiGLU blocks in the AR model.","blocks":[{"id":"n1","label":"LayerNorm"},{"id":"attn","label":"Bidirectional self-attention"},{"id":"r1","label":"Residual addition"},{"id":"n2","label":"LayerNorm"},{"id":"ffn","label":"Linear + GELU + linear"},{"id":"r2","label":"Residual addition"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ffn"],["ffn","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"sopro-speaker":{"title":"Sopro speaker and style conditioning","scope":"Separate identity and style heads share a convolutional trunk. Identity uses attention-weighted mean/std pooling; style pools statistics at multiple smoothing scales.","blocks":[{"id":"mel","label":"Speaker log-mel frontend"},{"id":"stem","label":"Conv1D + GroupNorm + SiLU"},{"id":"trunk","label":"Multiscale gated residual CNN","expand":"sopro-speaker-block"},{"id":"id","label":"Attentive statistics pooling + identity head"},{"id":"style","label":"Multiscale statistics + style/control heads"},{"id":"cond","label":"Speaker / style conditioning vector"}],"edges":[["mel","stem"],["stem","trunk"],["trunk","id"],["trunk","style"],["id","cond"],["style","cond"]]},"sopro-speaker-block":{"title":"Gated speaker CNN block","scope":"GroupNorm and sigmoid channel gating surround a dilated depthwise convolution. Squeeze-excitation supplies utterance-level channel weighting.","blocks":[{"id":"norm","label":"GroupNorm + pointwise projection"},{"id":"glu","label":"Sigmoid GLU"},{"id":"dw","label":"Dilated depthwise convolution"},{"id":"norm2","label":"GroupNorm + SiLU"},{"id":"se","label":"Squeeze-excitation channel gate"},{"id":"out","label":"Pointwise projection + residual"}],"edges":[["norm","glu"],["glu","dw"],["dw","norm2"],["norm2","se"],["se","out"]],"residuals":[["norm","out"]]},"sopro-ar":{"title":"Sopro semantic autoregression","scope":"Learned style queries attend to reference semantic embeddings. These style tokens, text embeddings and semantic prompt tokens form the prefix. This is a Sopro-trained decoder, not a pretrained Qwen checkpoint.","blocks":[{"id":"style","label":"Learned-query reference style pooling"},{"id":"prefix","label":"Style + text + semantic prompt embeddings"},{"id":"tr","label":"Causal rotary Transformer stack","expand":"sopro-ar-block"},{"id":"head","label":"RMSNorm + semantic token head"},{"id":"sample","label":"Token sampling / end prediction"}],"edges":[["style","prefix"],["prefix","tr"],["tr","head"],["head","sample"]]},"sopro-ar-block":{"title":"Sopro AR Transformer block","scope":"RMSNorm, rotary grouped-query attention and SwiGLU with learned residual scaling. QK RMS normalization is checkpoint-configured.","blocks":[{"id":"n1","label":"RMSNorm"},{"id":"attn","label":"Causal RoPE GQA / configured QK norm"},{"id":"r1","label":"Learned scale + residual"},{"id":"n2","label":"RMSNorm"},{"id":"ffn","label":"SwiGLU feed-forward"},{"id":"r2","label":"Learned scale + residual"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ffn"],["ffn","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"sopro-flow":{"title":"Sopro semantic-to-mel flow","scope":"Semantic embeddings are expanded to mel-frame rate. Reference mel and its mask anchor the prompt region; speaker/style vectors condition the generated continuation. The published audio.cpp route processes a complete text segment before emitting audio.","blocks":[{"id":"sem","label":"Semantic embedding + convolutional upsampling"},{"id":"concat","label":"Concat noisy mel, prompt mel, semantic and speaker features"},{"id":"pos","label":"Input / mask projection + convolutional positions"},{"id":"time","label":"Time embedding"},{"id":"tr","label":"Adaptive-LayerNorm DiT stack","expand":"sopro-flow-block"},{"id":"head","label":"Adaptive normalization + mel velocity head"},{"id":"solver","label":"Flow integration + prompt preservation"}],"edges":[["sem","concat"],["concat","pos"],["pos","tr"],["time","tr"],["tr","head"],["head","solver"]]},"sopro-flow-block":{"title":"Sopro acoustic DiT block","scope":"Time-derived shifts, scales and residual gates modulate attention and feed-forward branches. Rotary attention operates on acoustic frames; the feed-forward uses GELU, not SwiGLU.","blocks":[{"id":"n1","label":"Time-modulated LayerNorm"},{"id":"attn","label":"RoPE acoustic self-attention"},{"id":"r1","label":"Time gate + residual"},{"id":"n2","label":"Time-modulated LayerNorm"},{"id":"ffn","label":"Linear + GELU + linear"},{"id":"r2","label":"Time gate + residual"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ffn"],["ffn","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"echo-text":{"title":"Echo text conditioning","scope":"Byte IDs, not BPE or phonemes. A learned bidirectional text encoder produces per-layer conditioning K/V for the denoiser.","blocks":[{"id":"norm","label":"Punctuation / speaker-tag normalization"},{"id":"bytes","label":"BOS + UTF-8 byte IDs"},{"id":"emb","label":"Text embedding"},{"id":"tr","label":"Bidirectional Transformer stack","expand":"echo-encoder-block"},{"id":"kv","label":"RMSNorm + conditioning K/V projections"}],"edges":[["norm","bytes"],["bytes","emb"],["emb","tr"],["tr","kv"]]},"echo-reference":{"title":"Echo speaker conditioning","scope":"The reference transcript is not needed. The speaker encoder uses causal attention, while the text encoder is bidirectional. Both use the same block structure with separate parameters.","blocks":[{"id":"patch","label":"Fold groups of reference latent frames"},{"id":"proj","label":"Linear projection + activation scaling"},{"id":"tr","label":"Causal Transformer stack","expand":"echo-encoder-block"},{"id":"kv","label":"RMSNorm + speaker K/V projections"}],"edges":[["patch","proj"],["proj","tr"],["tr","kv"]]},"echo-encoder-block":{"title":"Echo conditioning Transformer block","scope":"Pre-RMSNorm, QK normalization, rotary positions, sigmoid-gated attention and SwiGLU. Causality depends on the conditioning branch.","blocks":[{"id":"norm","label":"RMSNorm"},{"id":"attn","label":"QK RMSNorm + RoPE self-attention"},{"id":"gate","label":"Sigmoid output gate + projection"},{"id":"add","label":"Residual addition"},{"id":"norm2","label":"RMSNorm"},{"id":"ffn","label":"SwiGLU feed-forward"},{"id":"add2","label":"Residual addition"}],"edges":[["norm","attn"],["attn","gate"],["gate","add"],["add","norm2"],["norm2","ffn"],["ffn","add2"]],"residuals":[["norm","add"],["norm2","add2"]]},"echo-dit":{"title":"Echo conditional diffusion","scope":"The audio.cpp route caches text and speaker K/V once, then samples the acoustic window. Its joint attention is noncausal within that window. Time conditioning uses low-rank adaptive RMS normalization, despite the upstream AdaLN class name.","blocks":[{"id":"noise","label":"Noisy PCA-space latent projection"},{"id":"time","label":"Sinusoidal time embedding + MLP"},{"id":"kv","label":"Text + speaker K/V","kind":"input"},{"id":"stack","label":"Joint-attention DiT stack","expand":"echo-dit-block"},{"id":"head","label":"Output normalization + latent prediction"},{"id":"sample","label":"Text / speaker CFG + iterative sampler"}],"edges":[["noise","stack"],["time","stack"],["kv","stack"],["stack","head"],["head","sample"]]},"echo-dit-block":{"title":"Echo joint-attention block","scope":"Self, text and speaker keys share one attention operation. RoPE applies to half of the self-attention heads, not half of every head's channels. Reference/text keys are normalized separately.","blocks":[{"id":"ada1","label":"Time-conditioned low-rank adaptive RMSNorm"},{"id":"attn","label":"Joint self / text / speaker attention"},{"id":"gate","label":"Sigmoid attention gate + output projection"},{"id":"res1","label":"Time gate + residual"},{"id":"ada2","label":"Time-conditioned low-rank adaptive RMSNorm"},{"id":"ffn","label":"SwiGLU feed-forward"},{"id":"res2","label":"Time gate + residual"}],"edges":[["ada1","attn"],["attn","gate"],["gate","res1"],["res1","ada2"],["ada2","ffn"],["ffn","res2"]],"residuals":[["ada1","res1"],["ada2","res2"]]},"echo-pca":{"title":"Echo codec-space transform","scope":"Encoding subtracts the stored mean, projects onto PCA components and scales the latent. Decoding reverses scaling, applies the transposed projection and restores the mean. This is a fixed linear transform, not a learned neural encoder.","blocks":[{"id":"center","label":"Center / restore codec mean"},{"id":"project","label":"PCA projection / inverse projection"},{"id":"scale","label":"Latent scaling / inverse scaling"}],"edges":[["center","project"],["project","scale"]]},"fish-dac-encoder":{"title":"Fish DAC reference encoding","scope":"Both integer codes and their reconstructed quantized latent can be obtained. Fish takes code IDs for AR prompting; Echo takes the latent and uses its own codec checkpoint.","blocks":[{"id":"cnn","label":"Causal Snake residual downsampling CNN","expand":"sam-dac-residual"},{"id":"attn","label":"Windowed Transformer in encoder","expand":"fish-codec-block"},{"id":"down","label":"ConvNeXt downsampling stages","expand":"midasheng-convnext"},{"id":"pre","label":"Windowed pre-quantizer Transformer","expand":"fish-codec-block"},{"id":"rvq","label":"Semantic + residual vector quantization"},{"id":"latent","label":"Sum projected codebook embeddings"}],"edges":[["cnn","attn"],["attn","down"],["down","pre"],["pre","rvq"],["rvq","latent"]]},"fish-dac-latent-decoder":{"title":"Fish DAC continuous-latent decoding","scope":"The entry point is the reconstructed quantizer latent. Echo supplies this directly after inverse PCA; no generated discrete-code lookup is required.","blocks":[{"id":"post","label":"Windowed post-quantizer Transformer"},{"id":"up","label":"ConvNeXt + transposed-convolution rate conversion"},{"id":"wave","label":"Snake transposed-convolution / residual stages"},{"id":"out","label":"Waveform convolution + tanh"}],"edges":[["post","up"],["up","wave"],["wave","out"]]},"omnivoice-diffusion":{"title":"OmniVoice discrete diffusion decoding","scope":"Target audio positions start masked. Each iteration predicts all remaining positions and commits a scheduled subset using confidence and codebook-level penalties. This is neither causal AR decoding nor continuous latent flow matching.","blocks":[{"id":"text","label":"Text / style token embeddings"},{"id":"audio","label":"Sum reference / masked codebook embeddings"},{"id":"pack","label":"Pack conditioning + target positions"},{"id":"qwen","label":"Noncausal Qwen3 Transformer stack","expand":"omnivoice-qwen3"},{"id":"heads","label":"Parallel audio-codebook logits"},{"id":"cfg","label":"Conditional / unconditional guidance"},{"id":"select","label":"Confidence selection + unmask schedule"},{"id":"codes","label":"Completed speech codes","kind":"output"}],"edges":[["text","pack"],["audio","pack"],["pack","qwen"],["qwen","heads"],["heads","cfg"],["cfg","select"],["select","codes"]]},"omnivoice-qwen3":{"title":"Qwen3 block for masked diffusion","scope":"Qwen3 normalization, rotary positions and SwiGLU are retained. Attention is not causal: available conditioning and target tokens are visible together. The generator re-evaluates the sequence as masked tokens are filled.","blocks":[{"id":"norm1","label":"RMSNorm"},{"id":"qkv","label":"Q / K / V projections"},{"id":"qknorm","label":"Per-head QK RMSNorm + RoPE"},{"id":"attn","label":"Noncausal grouped-query attention"},{"id":"out","label":"Output projection + residual"},{"id":"norm2","label":"RMSNorm"},{"id":"ffn","label":"SwiGLU feed-forward"},{"id":"add","label":"Residual addition"}],"edges":[["norm1","qkv"],["qkv","qknorm"],["qknorm","attn"],["attn","out"],["out","norm2"],["norm2","ffn"],["ffn","add"]],"residuals":[["norm1","out"],["norm2","add"]]},"higgs-v2-reference":{"title":"Higgs Audio V2 dual-branch encoder","scope":"The semantic branch averages HuBERT hidden states, aligns their frame rate and applies a convolutional encoder. The acoustic branch separately downsamples the waveform. Residual quantization encodes their fused representation.","blocks":[{"id":"hubert","label":"HuBERT waveform CNN + Transformer","expand":"higgs-v2-hubert"},{"id":"semantic","label":"Hidden-state mean + temporal alignment + CNN"},{"id":"acoustic","label":"Snake residual CNN downsampling"},{"id":"fuse","label":"Concatenate + linear projection"},{"id":"rvq","label":"Residual vector quantization"},{"id":"codes","label":"Reference speech codes","kind":"output"}],"edges":[["hubert","semantic"],["semantic","fuse"],["acoustic","fuse"],["fuse","rvq"],["rvq","codes"]]},"higgs-v2-hubert":{"title":"HuBERT semantic branch","scope":"Unlike selecting a single content layer, this tokenizer averages the input representation and Transformer layer outputs before temporal alignment.","blocks":[{"id":"cnn","label":"Waveform convolution feature extractor"},{"id":"project","label":"Normalize + feature projection"},{"id":"pos","label":"Grouped positional convolution + normalization"},{"id":"tr","label":"Post-norm Transformer stack","expand":"hubert-layer"},{"id":"mean","label":"Mean across hidden-state levels"}],"edges":[["cnn","project"],["project","pos"],["pos","tr"],["pos","mean"],["tr","mean"]]},"higgs-v2-decoder":{"title":"Higgs Audio V2 waveform decoding","scope":"Codes reconstruct the fused latent. Waveform synthesis uses the acoustic decoder; it does not run the HuBERT reference encoder.","blocks":[{"id":"lookup","label":"Codebook lookup + per-codebook projection"},{"id":"sum","label":"Sum quantized residual contributions"},{"id":"project","label":"Acoustic latent projection + convolution"},{"id":"up","label":"Snake + transposed convolution stages"},{"id":"res","label":"Dilated Snake residual units"},{"id":"wave","label":"Snake + waveform convolution"}],"edges":[["lookup","sum"],["sum","project"],["project","up"],["up","res"],["res","wave"]]},"omnivoice-duration":{"title":"Rule-based audio length","scope":"Reference text and duration calibrate speaking rate when available; otherwise the estimator uses a fallback rate. The request may explicitly provide a duration. No duration-network weights are involved.","blocks":[{"id":"text","label":"Language-aware text weights"},{"id":"ref","label":"Optional reference duration / transcript"},{"id":"rate","label":"Estimate speaking rate"},{"id":"speed","label":"Apply requested speed"},{"id":"frames","label":"Allocate target code frames"}],"edges":[["text","rate"],["ref","rate"],["rate","speed"],["speed","frames"]]},"voxcpm-minicpm":{"title":"MiniCPM Transformer block","scope":"Shared block structure, not shared weights. Temporal language models use causal attention and KV state; local patch encoding and diffusion use noncausal attention. MiniCPM depth scaling is configuration-dependent.","blocks":[{"id":"n1","label":"RMSNorm"},{"id":"attn","label":"RoPE grouped-query attention"},{"id":"r1","label":"Scaled residual addition"},{"id":"n2","label":"RMSNorm"},{"id":"ffn","label":"SwiGLU feed-forward"},{"id":"r2","label":"Scaled residual addition"}],"edges":[["n1","attn"],["attn","r1"],["r1","n2"],["n2","ffn"],["ffn","r2"]],"residuals":[["n1","r1"],["n2","r2"]]},"voxcpm-local":{"title":"Local acoustic patch encoder","scope":"One summary embedding per continuous audio patch. This is not a vector-quantized speech tokenizer.","blocks":[{"id":"proj","label":"Latent patch projection"},{"id":"cls","label":"Prepend learned summary token"},{"id":"tr","label":"Noncausal MiniCPM stack","expand":"voxcpm-minicpm"},{"id":"select","label":"Select summary token"},{"id":"out","label":"Project to language-model width"}],"edges":[["proj","cls"],["cls","tr"],["tr","select"],["select","out"]]},"voxcpm-ar":{"title":"Hierarchical temporal AR conditioning","scope":"Text embeddings and encoded acoustic patches form the prompt. Each generated patch is locally encoded for the next AR step. V1 adds acoustic and semantic embeddings for the residual LM; V2 concatenates and projects them. The semantic bottleneck uses finite scalar quantization, not codec IDs.","blocks":[{"id":"prompt","label":"Text / acoustic patch embeddings","kind":"input"},{"id":"base","label":"Text-semantic MiniCPM AR","expand":"voxcpm-minicpm"},{"id":"fsq","label":"Projection + tanh + scalar rounding + projection"},{"id":"fuse","label":"Fuse semantic state with patch embedding"},{"id":"res","label":"Residual acoustic MiniCPM AR","expand":"voxcpm-minicpm"},{"id":"sem","label":"Semantic conditioning projection"},{"id":"ac","label":"Acoustic conditioning projection"},{"id":"stop","label":"Stop prediction head"}],"edges":[["prompt","base"],["base","fsq"],["fsq","fuse"],["prompt","fuse"],["fuse","res"],["fsq","sem"],["res","ac"],["fsq","stop"]]},"voxcpm1-flow":{"title":"VoxCPM1 local diffusion","scope":"The AR semantic and residual projections are summed, then added to the timestep embedding. Flow integration generates continuous latent patches. This is token-prefix conditioning, not an AdaLN DiT.","blocks":[{"id":"mu","label":"Sum semantic + acoustic conditioning"},{"id":"time","label":"Time / delta-time MLPs"},{"id":"sum","label":"Add conditioning to time token"},{"id":"patch","label":"Previous + noisy patch projections"},{"id":"seq","label":"Concatenate time and patch tokens"},{"id":"tr","label":"Noncausal MiniCPM stack","expand":"voxcpm-minicpm"},{"id":"head","label":"Select noisy positions + velocity projection"},{"id":"solve","label":"Flow integration"}],"edges":[["mu","sum"],["time","sum"],["sum","seq"],["patch","seq"],["seq","tr"],["tr","head"],["head","solve"]]},"voxcpm2-flow":{"title":"VoxCPM2 local diffusion","scope":"Separate semantic and acoustic tokens preserve their identities in the diffusion prefix. Time and delta-time are embedded by MLPs. Generated patches recur through the local encoder at the next AR step.","blocks":[{"id":"mu","label":"Semantic + acoustic prefix tokens"},{"id":"time","label":"Time / delta-time MLPs"},{"id":"patch","label":"Previous + noisy patch projections"},{"id":"seq","label":"Concatenate conditioning, time and patches"},{"id":"tr","label":"Noncausal MiniCPM stack","expand":"voxcpm-minicpm"},{"id":"head","label":"Select noisy positions + velocity projection"},{"id":"solve","label":"Flow integration"}],"edges":[["mu","seq"],["time","seq"],["patch","seq"],["seq","tr"],["tr","head"],["head","solve"]]},"voxcpm-vae-encoder":{"title":"AudioVAE continuous encoding","scope":"AudioVAE V1/V2 have distinct weights and rate configurations. Reference inference uses the posterior mean; no waveform-codebook lookup is involved.","blocks":[{"id":"conv","label":"Causal input convolution"},{"id":"res","label":"Snake + dilated residual convolutions"},{"id":"down","label":"Strided convolution stages"},{"id":"mean","label":"Latent mean projection"},{"id":"patch","label":"Group frames into acoustic patches"}],"edges":[["conv","res"],["res","down"],["down","mean"],["mean","patch"]]},"voxcpm-vae-decoder":{"title":"AudioVAE continuous decoding","scope":"Each upsampling stage combines Snake activation, transposed convolution and dilated residual units. Output rate and upsampling strides belong to the checkpoint configuration; V2 uses asymmetric encode/decode rates.","blocks":[{"id":"input","label":"Continuous latent convolution"},{"id":"up","label":"Snake + transposed convolution"},{"id":"res","label":"Dilated Snake residual convolution units"},{"id":"repeat","label":"Repeat upsampling stages"},{"id":"out","label":"Snake + causal waveform convolution"}],"edges":[["input","up"],["up","res"],["res","repeat"],["repeat","out"]]},"pocket-text":{"title":"PocketTTS text conditioning","scope":"Text tokens are embedded directly for the AR prompt; there is no separate pretrained text encoder or phonemizer.","blocks":[{"id":"normalize","label":"Language-specific text normalization"},{"id":"tokens","label":"SentencePiece tokenization"},{"id":"embed","label":"Learned token embeddings"}],"edges":[["normalize","tokens"],["tokens","embed"]]},"pocket-reference":{"title":"Continuous Mimi reference encoding","scope":"Reference audio can be encoded into a FlowLM prefix, or a previously exported voice state can bypass this computation. A reference transcript is not required.","blocks":[{"id":"cnn","label":"Causal SEANet residual downsampling"},{"id":"transformer","label":"Causal rotary Transformer stack","expand":"pocket-transformer"},{"id":"down","label":"Frame-rate downsampling convolution"},{"id":"project","label":"Continuous-latent speaker projection"},{"id":"prompt","label":"Voice prefix / optional BOS embedding"}],"edges":[["cnn","transformer"],["transformer","down"],["down","project"],["project","prompt"]]},"pocket-flowlm":{"title":"FlowLM autoregressive conditioning","scope":"The flow head produces the next latent; that latent is projected back into FlowLM at the next acoustic step. This recurrence is described here rather than drawn as a cycle in the pipeline DAG.","blocks":[{"id":"text","label":"Text / voice prefix embeddings","kind":"input"},{"id":"latent","label":"Previous latent / BOS projection"},{"id":"stack","label":"Causal rotary Transformer stack","expand":"pocket-transformer"},{"id":"norm","label":"Final LayerNorm"},{"id":"condition","label":"Flow-head conditioning","kind":"output"},{"id":"stop","label":"Stop probability projection"}],"edges":[["text","stack"],["latent","stack"],["stack","norm"],["norm","condition"],["norm","stop"]]},"pocket-transformer":{"title":"PocketTTS streaming Transformer block","scope":"Pre-normalized causal attention with rotary positions and LayerScale residual branches. Bounded attention state is retained for streaming; FlowLM and Mimi have separate weights and dimensions.","blocks":[{"id":"norm1","label":"LayerNorm"},{"id":"attention","label":"Causal RoPE attention + KV state"},{"id":"add1","label":"LayerScale + residual"},{"id":"norm2","label":"LayerNorm"},{"id":"ffn","label":"Feed-forward network"},{"id":"add2","label":"LayerScale + residual"}],"edges":[["norm1","attention"],["attention","add1"],["add1","norm2"],["norm2","ffn"],["ffn","add2"]],"residuals":[["norm1","add1"],["norm2","add2"]]},"pocket-flow-head":{"title":"Time-conditioned flow MLP","scope":"This head operates on one acoustic latent at a time. It is a residual MLP, not a sequence-level Diffusion Transformer.","blocks":[{"id":"noise","label":"Noise projection"},{"id":"condition","label":"AR condition + start/end time embeddings"},{"id":"res","label":"AdaLN residual MLP layers"},{"id":"head","label":"Final AdaLN / latent projection"},{"id":"step","label":"Flow update / next latent"}],"edges":[["noise","res"],["condition","res"],["res","head"],["condition","head"],["head","step"]]},"pocket-mimi-decoder":{"title":"Mimi continuous waveform decoding","scope":"PocketTTS uses continuous latent input. The decoder retains convolution and attention state across emitted chunks, including overlap from transposed convolutions.","blocks":[{"id":"project","label":"Continuous latent projection"},{"id":"rate","label":"Depthwise rate upsampling"},{"id":"transformer","label":"Causal rotary Transformer stack","expand":"pocket-transformer"},{"id":"conv","label":"Input convolution"},{"id":"up","label":"ELU / transposed-convolution stages"},{"id":"res","label":"Causal residual convolution units"},{"id":"out","label":"Output waveform convolution"}],"edges":[["project","rate"],["rate","transformer"],["transformer","conv"],["conv","up"],["up","res"],["res","out"]]},"neutts-preset":{"title":"Packaged speaker conditioning","scope":"The 2E package uses stored transcript/code pairs. Emotion is a prompt token, not a separate emotion encoder.","blocks":[{"id":"voice","label":"Select speaker ID"},{"id":"lookup","label":"Load stored transcript + speech codes"},{"id":"text","label":"Combine reference / target text + emotion"},{"id":"prompt","label":"Assemble text / speech token prompt"}],"edges":[["voice","lookup"],["lookup","text"],["text","prompt"],["lookup","prompt"]]},"neucodec":{"title":"NeuCodec reconstruction","scope":"A single integer ID decodes into several finite scalar levels. The reconstructed continuous features feed a spectral decoder, not a learned RVQ lookup stack.","blocks":[{"id":"levels","label":"Unpack FSQ scalar levels"},{"id":"project","label":"Quantizer / acoustic projections"},{"id":"conv","label":"Input convolution + residual CNN"},{"id":"transformer","label":"Rotary Transformer stack","expand":"neucodec-layer"},{"id":"post","label":"Residual CNN + LayerNorm"},{"id":"spectrum","label":"Magnitude / phase projection"},{"id":"istft","label":"Inverse STFT / output resampling"}],"edges":[["levels","project"],["project","conv"],["conv","transformer"],["transformer","post"],["post","spectrum"],["spectrum","istft"]]},"neucodec-layer":{"title":"NeuCodec Transformer layer","scope":"RMS-normalized attention and an ordinary SiLU feed-forward branch. This decoder block is not the Qwen3 AR generator and does not use its SwiGLU feed-forward structure.","blocks":[{"id":"norm1","label":"RMSNorm"},{"id":"attention","label":"Rotary self-attention"},{"id":"add1","label":"Residual add"},{"id":"norm2","label":"RMSNorm"},{"id":"ffn","label":"Linear / SiLU / linear"},{"id":"add2","label":"Residual add"}],"edges":[["norm1","attention"],["attention","add1"],["add1","norm2"],["norm2","ffn"],["ffn","add2"]],"residuals":[["norm1","add1"],["norm2","add2"]]},"dac-encoder":{"title":"DAC reference tokenization","scope":"Projected normalized codebook matching quantizes the remaining residual at each stage. OuteTTS uses a two-codebook checkpoint; other DAC derivatives can differ.","blocks":[{"id":"conv","label":"Waveform input convolution"},{"id":"res","label":"Snake dilated residual units"},{"id":"down","label":"Strided convolution / repeat scales"},{"id":"latent","label":"Latent projection"},{"id":"vq","label":"Projected residual vector quantization"},{"id":"ids","label":"Codebook IDs","kind":"output"}],"edges":[["conv","res"],["res","down"],["down","latent"],["latent","vq"],["vq","ids"]]},"dac-decoder":{"title":"DAC waveform reconstruction","scope":"Each selected codebook embedding is projected to latent channels; their sum conditions convolutional waveform synthesis.","blocks":[{"id":"lookup","label":"Codebook lookup / output projections"},{"id":"sum","label":"Sum quantizer contributions"},{"id":"pre","label":"Latent input convolution"},{"id":"up","label":"Snake + transposed-convolution upsampling"},{"id":"res","label":"Dilated Snake residual units"},{"id":"out","label":"Repeat scales / output convolution / tanh"}],"edges":[["lookup","sum"],["sum","pre"],["pre","up"],["up","res"],["res","out"]]},"oute-aligner":{"title":"Reference alignment companion","scope":"This is the aligner selected by audio.cpp's cloning implementation, not a claim that the upstream OuteTTS model was trained with a Qwen3 component.","blocks":[{"id":"audio","label":"Qwen audio encoder","expand":"qwen-audio"},{"id":"text","label":"Transcript / timestamp prompt"},{"id":"backbone","label":"Qwen3 causal backbone","expand":"qwen3-layer"},{"id":"head","label":"Timestamp classification","expand":"qwen-align-head"}],"edges":[["audio","backbone"],["text","backbone"],["backbone","head"]]},"oute-profile":{"title":"OuteTTS cloning prompt","scope":"The reference transcript is aligned to codec frames. Word timing and acoustic statistics describe the reference alongside audio codes in the serialized prompt.","blocks":[{"id":"alignment","label":"Reference word spans","kind":"input"},{"id":"codes","label":"DAC reference codes","kind":"input"},{"id":"stats","label":"Energy / pitch / spectral statistics"},{"id":"group","label":"Group codes and statistics by word"},{"id":"prompt","label":"Serialize profile + target text"}],"edges":[["alignment","group"],["codes","group"],["stats","group"],["group","prompt"]]},"f5-tokenizer":{"title":"F5 / Habibi text input","scope":"The packaged Habibi path uses its character vocabulary and dialect formatting; other upstream F5 vocabularies are not interchangeable.","blocks":[{"id":"normalize","label":"Checkpoint text normalization"},{"id":"prompt","label":"Reference transcript + target text"},{"id":"dialect","label":"Habibi dialect / boundary tokens"},{"id":"ids","label":"Vocabulary IDs"}],"edges":[["normalize","prompt"],["prompt","dialect"],["dialect","ids"]]},"f5-text":{"title":"F5 ConvNeXt V2 text conditioning","scope":"Text positions are padded to the acoustic sequence length. Padding is masked through the text blocks; there is no separate phoneme-duration predictor.","blocks":[{"id":"embed","label":"Token embedding + sinusoidal positions"},{"id":"pad","label":"Pad / mask to acoustic length"},{"id":"conv","label":"Depthwise convolution + LayerNorm"},{"id":"mlp","label":"Pointwise expansion + GELU"},{"id":"grn","label":"Global-response normalization"},{"id":"out","label":"Pointwise projection + residual"}],"edges":[["embed","pad"],["pad","conv"],["conv","mlp"],["mlp","grn"],["grn","out"]],"residuals":[["conv","out"]]},"f5-dit":{"title":"F5 mel flow Transformer","scope":"Reference mel occupies the prompt portion of the sequence. Noisy mel and text conditioning are concatenated before projection; the generated continuation is cropped before vocoding.","blocks":[{"id":"input","label":"Text / noisy mel / reference mel projection"},{"id":"position","label":"Grouped convolution position embedding"},{"id":"time","label":"Time embedding MLP","kind":"input"},{"id":"layers","label":"Rotary DiT layers","expand":"f5-dit-layer"},{"id":"head","label":"Adaptive LayerNorm + mel velocity"},{"id":"integrate","label":"Guided flow integration / crop prompt"}],"edges":[["input","position"],["position","layers"],["time","layers"],["layers","head"],["time","head"],["head","integrate"]]},"f5-dit-layer":{"title":"F5 DiT layer","scope":"Time supplies shift, scale and gate vectors to both residual branches. The feed-forward branch uses GELU, not SwiGLU.","blocks":[{"id":"norm1","label":"Time-adaptive LayerNorm"},{"id":"attn","label":"Noncausal RoPE self-attention"},{"id":"add1","label":"Time gate + residual"},{"id":"norm2","label":"Time-adaptive LayerNorm"},{"id":"ffn","label":"Linear / GELU / linear"},{"id":"add2","label":"Time gate + residual"}],"edges":[["norm1","attn"],["attn","add1"],["add1","norm2"],["norm2","ffn"],["ffn","add2"]],"residuals":[["norm1","add1"],["norm2","add2"]]},"zipvoice-tokenizer":{"title":"Emilia text frontend","scope":"Chinese and English use different pronunciation frontends before mapping to the same checkpoint token vocabulary.","blocks":[{"id":"text","label":"Normalize / split language spans"},{"id":"zh","label":"Chinese segmentation + pinyin"},{"id":"en","label":"English eSpeak phonemes"},{"id":"ids","label":"Merge symbols / vocabulary lookup"}],"edges":[["text","zh"],["text","en"],["zh","ids"],["en","ids"]]},"zipvoice-text":{"title":"Zipformer text conditioning","scope":"Prompt and target token features are expanded using the reference-duration ratio and requested speed. It is not an autoregressive text decoder.","blocks":[{"id":"embed","label":"Prompt / target token embeddings"},{"id":"layers","label":"Zipformer text layers","expand":"zipvoice-layer"},{"id":"duration","label":"Reference duration / token-ratio estimate"},{"id":"expand","label":"Expand text features to acoustic frames"}],"edges":[["embed","layers"],["layers","expand"],["duration","expand"]]},"zipvoice-flow":{"title":"Zipformer flow decoder","scope":"Stacks operate at multiple temporal rates and restore the acoustic sequence length. Time conditioning enters the flow network; the text encoder uses the related block without time conditioning.","blocks":[{"id":"in","label":"Noisy mel / reference / text projection"},{"id":"time","label":"Time embedding MLP","kind":"input"},{"id":"down","label":"Learned temporal downsampling"},{"id":"layers","label":"Multi-rate Zipformer stacks","expand":"zipvoice-layer"},{"id":"up","label":"Upsampling / bypass combination"},{"id":"head","label":"Mel velocity projection"},{"id":"solve","label":"Guided flow integration / crop prompt"}],"edges":[["in","down"],["down","layers"],["time","layers"],["layers","up"],["in","up"],["up","head"],["head","solve"]]},"zipvoice-layer":{"title":"Zipformer layer","scope":"Attention weights are computed once and reused by nonlinear and value-attention branches. Swoosh feed-forward and gated depthwise-convolution branches use residual updates; learned bypasses mix the layer input back in.","blocks":[{"id":"weights","label":"Relative-position attention weights"},{"id":"ff1","label":"Feed-forward residual (SwooshL)"},{"id":"attn1","label":"Nonlinear + first value attention residuals"},{"id":"conv1","label":"Gated depthwise convolution + FFN residuals"},{"id":"bypass","label":"Mid-layer learned bypass"},{"id":"attn2","label":"Second value attention residual"},{"id":"conv2","label":"Convolution + third FFN residuals"},{"id":"out","label":"BiasNorm + output bypass"}],"edges":[["weights","attn1"],["ff1","attn1"],["attn1","conv1"],["conv1","bypass"],["bypass","attn2"],["weights","attn2"],["attn2","conv2"],["conv2","out"]]},"audiosr-frontend":{"title":"AudioSR bandwidth conditioning","scope":"The input is a recording, not a text prompt. Bandwidth analysis prepares the low-pass signal used by both conditioning and reconstruction.","blocks":[{"id":"prep","label":"Mono / 48 kHz / level preparation"},{"id":"cutoff","label":"STFT energy cutoff estimate"},{"id":"lowpass","label":"Low-pass filtering / resampling"},{"id":"mel","label":"Mel spectrogram"}],"edges":[["prep","cutoff"],["cutoff","lowpass"],["lowpass","mel"]]},"audiosr-vae-encoder":{"title":"Mel latent encoder","scope":"Residual 2D convolution levels reduce time-frequency resolution. The conditional encoder and reconstruction decoder have separate checkpoint prefixes.","blocks":[{"id":"conv","label":"Input 2D convolution"},{"id":"down","label":"Residual blocks + strided downsampling","expand":"audiosr-vae-residual"},{"id":"mid","label":"Residual / spatial attention / residual"},{"id":"stats","label":"GroupNorm / SiLU / moment projection"},{"id":"sample","label":"Gaussian latent sample + scale"}],"edges":[["conv","down"],["down","mid"],["mid","stats"],["stats","sample"]]},"audiosr-vae-decoder":{"title":"Mel latent reconstruction","scope":"The output is a spectrogram, not a waveform. HiFi-GAN performs the subsequent acoustic synthesis.","blocks":[{"id":"conv","label":"Latent projection + input convolution"},{"id":"mid","label":"Residual / spatial attention / residual"},{"id":"up","label":"Residual blocks + interpolation / convolution","expand":"audiosr-vae-residual"},{"id":"out","label":"GroupNorm / SiLU / mel projection"}],"edges":[["conv","mid"],["mid","up"],["up","out"]]},"audiosr-vae-residual":{"title":"Mel VAE residual block","scope":"Group-normalized SiLU convolutions with identity or projected channel shortcut.","blocks":[{"id":"norm1","label":"GroupNorm + SiLU"},{"id":"conv1","label":"3 × 3 convolution"},{"id":"norm2","label":"GroupNorm + SiLU"},{"id":"conv2","label":"3 × 3 convolution"},{"id":"add","label":"Projected / identity residual add"}],"edges":[["norm1","conv1"],["conv1","norm2"],["norm2","conv2"],["conv2","add"]],"residuals":[["norm1","add"]]},"audiosr-unet":{"title":"AudioSR latent diffusion","scope":"The condition latent is concatenated with noisy latents. Spatial Transformer attention uses the current feature sequence, not an external language encoder. DDIM updates the latent over diffusion steps.","blocks":[{"id":"cat","label":"Noisy latent + condition concatenation"},{"id":"time","label":"Timestep embedding MLP","kind":"input"},{"id":"down","label":"Residual / Transformer down path","expand":"audiosr-spatial"},{"id":"mid","label":"Residual / Transformer bottleneck","expand":"audiosr-spatial"},{"id":"up","label":"Skip concatenation + up path","expand":"audiosr-spatial"},{"id":"head","label":"Normalized convolution noise prediction"},{"id":"step","label":"CFG / DDIM update"}],"edges":[["cat","down"],["time","down"],["down","mid"],["time","mid"],["mid","up"],["down","up"],["time","up"],["up","head"],["head","step"]]},"audiosr-spatial":{"title":"Spatial Transformer sub-block","scope":"2D features are normalized and projected to a token sequence. Two self-attention residuals and a GEGLU feed-forward residual precede spatial projection back to the U-Net.","blocks":[{"id":"pre","label":"GroupNorm / projection / flatten"},{"id":"attn1","label":"LayerNorm + self-attention"},{"id":"add1","label":"Residual add"},{"id":"attn2","label":"LayerNorm + self-attention"},{"id":"add2","label":"Residual add"},{"id":"ff","label":"LayerNorm + GEGLU FFN"},{"id":"add3","label":"Residual add"},{"id":"post","label":"Spatial projection + outer residual"}],"edges":[["pre","attn1"],["attn1","add1"],["add1","attn2"],["attn2","add2"],["add2","ff"],["ff","add3"],["add3","post"]],"residuals":[["attn1","add1"],["attn2","add2"],["ff","add3"],["pre","post"]]},"hifigan":{"title":"HiFi-GAN waveform generator","scope":"Unlike ISTFT vocoders, the output convolution directly predicts waveform samples. Kernel sizes, channel widths and upsampling factors vary by checkpoint.","blocks":[{"id":"pre","label":"Mel input convolution"},{"id":"up","label":"LeakyReLU + transposed convolution"},{"id":"banks","label":"Parallel dilated residual convolutions"},{"id":"avg","label":"Average banks / repeat upsampling"},{"id":"out","label":"LeakyReLU / convolution / tanh"}],"edges":[["pre","up"],["up","banks"],["banks","avg"],["avg","out"]]},"audiosr-preserve":{"title":"Low-frequency preservation","scope":"These are two separate points in the pipeline: mel replacement before vocoding, and reference STFT replacement after it. The diagram summarizes their order without introducing an extra model.","blocks":[{"id":"mel","label":"Replace generated low mel bins"},{"id":"vocoder","label":"HiFi-GAN synthesis","expand":"hifigan"},{"id":"stft","label":"Match cutoff energy / restore low STFT bins"},{"id":"istft","label":"Inverse STFT / normalize / trim"}],"edges":[["mel","vocoder"],["vocoder","stft"],["stft","istft"]]},"campplus":{"title":"CAMPPlus reference encoding","scope":"CAMPPlus uses context-aware masking in densely connected TDNN blocks. It is not ECAPA-TDNN or an audio tokenizer.","blocks":[{"id":"front","label":"Residual 2D convolution frontend"},{"id":"tdnn","label":"Temporal convolution / downsampling"},{"id":"dense","label":"Dense CAM-TDNN blocks","expand":"campplus-block"},{"id":"pool","label":"Temporal mean / standard deviation"},{"id":"out","label":"Embedding projection / normalization"}],"edges":[["front","tdnn"],["tdnn","dense"],["dense","pool"],["pool","out"]]},"campplus-block":{"title":"Context-aware masking TDNN","scope":"Local and global context generate a sigmoid mask. Dense concatenation, rather than residual addition, retains earlier features.","blocks":[{"id":"project","label":"Normalize / activate / bottleneck"},{"id":"local","label":"Local temporal convolution"},{"id":"context","label":"Segment + global average context"},{"id":"gate","label":"Pointwise network + sigmoid"},{"id":"mask","label":"Mask local features"},{"id":"concat","label":"Concatenate with incoming features"}],"edges":[["project","local"],["project","context"],["context","gate"],["gate","mask"],["local","mask"],["mask","concat"]]},"ssl-content":{"title":"Wav2Vec2-family content features","scope":"HuBERT and XLS-R share this broad waveform-CNN/Transformer layout, but not their training objective or weights. Normalization placement and selected hidden layer follow the checkpoint.","blocks":[{"id":"cnn","label":"Strided waveform CNN + GELU"},{"id":"project","label":"Feature normalization / projection"},{"id":"pos","label":"Grouped positional convolution"},{"id":"layers","label":"Bidirectional self-attention + GELU FFN"},{"id":"select","label":"Select configured hidden layer"}],"edges":[["cnn","project"],["project","pos"],["pos","layers"],["layers","select"]]},"seed-astral":{"title":"HuBERT / ASTRAL content tokens","scope":"The current v2 conversion route uses the wide ASTRAL quantizer for source and reference. Binary spherical quantization uses sign patterns, not a learned nearest-neighbor codebook.","blocks":[{"id":"ssl","label":"HuBERT-Large hidden features","expand":"ssl-content"},{"id":"project","label":"Feature projection"},{"id":"conv","label":"ConvNeXt + global-response normalization"},{"id":"latent","label":"Quantizer projection"},{"id":"bsq","label":"Binary spherical quantization"},{"id":"ids","label":"Content token IDs","kind":"output"}],"edges":[["ssl","project"],["project","conv"],["conv","latent"],["latent","bsq"],["bsq","ids"]]},"seed-length":{"title":"Content-to-mel-frame conditioning","scope":"v1 projects continuous features; v2 embeds discrete tokens. Optional pitch enters the v1 singing route. This aligns an existing sequence to a requested frame length, not AR duration prediction.","blocks":[{"id":"project","label":"Feature projection / token embedding"},{"id":"resize","label":"Nearest frame-length interpolation"},{"id":"pitch","label":"Optional pitch embedding","kind":"input"},{"id":"sum","label":"Combine frame conditions"},{"id":"conv","label":"Convolution + GroupNorm + Mish stages"},{"id":"out","label":"Frame-aligned condition","kind":"output"}],"edges":[["project","resize"],["resize","sum"],["pitch","sum"],["sum","conv"],["conv","out"]]},"seed-uvit":{"title":"Seed-VC v1 U-ViT flow","scope":"Content, noisy mel and reference mel are projected jointly. Style/time enter as tokens or conditioning according to checkpoint. Encoder-side states are concatenated into later layers through U-shaped skips.","blocks":[{"id":"in","label":"Mel / content / prompt projection"},{"id":"cond","label":"Time + speaker conditioning","kind":"input"},{"id":"early","label":"Early rotary Transformer layers","expand":"seed-uvit-layer"},{"id":"middle","label":"Middle Transformer layer","expand":"seed-uvit-layer"},{"id":"late","label":"Skip concatenation + late layers","expand":"seed-uvit-layer"},{"id":"head","label":"MLP or WaveNet velocity head"},{"id":"solve","label":"Guided flow integration"}],"edges":[["in","early"],["cond","early"],["early","middle"],["middle","late"],["early","late"],["late","head"],["head","solve"]]},"seed-uvit-layer":{"title":"U-ViT Transformer layer","scope":"Noncausal rotary attention and SwiGLU feed-forward blocks. Checkpoints using time as a token use ordinary RMSNorm; others use time-adaptive RMSNorm.","blocks":[{"id":"norm1","label":"RMSNorm / time-adaptive RMSNorm"},{"id":"attn","label":"RoPE self-attention"},{"id":"add1","label":"Residual add"},{"id":"norm2","label":"RMSNorm / time-adaptive RMSNorm"},{"id":"ffn","label":"SwiGLU feed-forward"},{"id":"add2","label":"Residual add"}],"edges":[["norm1","attn"],["attn","add1"],["add1","norm2"],["norm2","ffn"],["ffn","add2"]],"residuals":[["norm1","add1"],["norm2","add2"]]},"seed-v2-cfm":{"title":"Seed-VC v2 conditional flow","scope":"The v2 CFM path directly consumes length-regulated ASTRAL tokens. Separate content/style guidance controls do not make it an autoregressive decoder.","blocks":[{"id":"in","label":"Noisy mel + content + reference projection"},{"id":"cond","label":"Time / speaker conditioning","kind":"input"},{"id":"layers","label":"Adaptive rotary Transformer stack","expand":"seed-v2-layer"},{"id":"head","label":"Adaptive normalization + velocity head"},{"id":"solve","label":"Guided flow integration"}],"edges":[["in","layers"],["cond","layers"],["layers","head"],["cond","head"],["head","solve"]]},"seed-v2-layer":{"title":"Seed-VC v2 DiT layer","scope":"Time conditioning supplies shifts, scales and residual gates for both attention and feed-forward branches.","blocks":[{"id":"norm1","label":"Time-modulated RMSNorm"},{"id":"attn","label":"RoPE self-attention"},{"id":"add1","label":"Time gate + residual"},{"id":"norm2","label":"Time-modulated RMSNorm"},{"id":"ffn","label":"SwiGLU feed-forward"},{"id":"add2","label":"Time gate + residual"}],"edges":[["norm1","attn"],["attn","add1"],["add1","norm2"],["norm2","ffn"],["ffn","add2"]],"residuals":[["norm1","add1"],["norm2","add2"]]},"bigvgan":{"title":"BigVGAN waveform synthesis","scope":"The generator upsamples acoustic frames directly to waveform samples. It does not use an inverse-STFT output head.","blocks":[{"id":"pre","label":"Acoustic input convolution"},{"id":"up","label":"Transposed-convolution upsampling"},{"id":"res","label":"Parallel dilated residual banks","expand":"bigvgan-residual"},{"id":"repeat","label":"Average banks / repeat scales"},{"id":"post","label":"Periodic activation + output convolution"},{"id":"out","label":"Waveform bounding","kind":"output"}],"edges":[["pre","up"],["up","res"],["res","repeat"],["repeat","post"],["post","out"]]},"bigvgan-residual":{"title":"Anti-aliased periodic residual block","scope":"Low-pass resampling surrounds Snake / SnakeBeta nonlinearities to limit aliases introduced by periodic activation. Multiple dilation branches are aggregated by the generator.","blocks":[{"id":"act1","label":"Upsample / periodic activation / downsample"},{"id":"conv1","label":"Dilated convolution"},{"id":"act2","label":"Anti-aliased periodic activation"},{"id":"conv2","label":"Convolution"},{"id":"add","label":"Residual add"}],"edges":[["act1","conv1"],["conv1","act2"],["act2","conv2"],["conv2","add"]],"residuals":[["act1","add"]]},"hift":{"title":"HiFT harmonic ISTFT synthesis","scope":"An internal pitch predictor supplies harmonic excitation. This pitch prediction is distinct from an optional external RMVPE branch used to condition the upstream flow model.","blocks":[{"id":"mel","label":"Mel features"},{"id":"conv","label":"Input convolution"},{"id":"f0","label":"Convolutional F0 prediction"},{"id":"source","label":"Harmonic / noise source + STFT"},{"id":"up","label":"Upsampling + source injection"},{"id":"res","label":"Periodic residual blocks"},{"id":"spec","label":"Magnitude / phase prediction"},{"id":"istft","label":"Inverse STFT"}],"edges":[["mel","conv"],["mel","f0"],["conv","up"],["f0","source"],["source","up"],["up","res"],["res","spec"],["spec","istft"]]},"coco-content-style":{"title":"CoCo content-style encoding","scope":"Whisper features are combined with chromagram features before the tokenizer's downsampling and ConvNeXt stack. The single-codebook output runs at 12.5 Hz in the published Vevo2 checkpoint.","blocks":[{"id":"chroma","label":"Chromagram frontend"},{"id":"whisper","label":"Whisper audio encoder","expand":"whisper-encoder"},{"id":"sum","label":"Branch convolution projections + sum"},{"id":"down","label":"Strided convolutions + GELU"},{"id":"convnext","label":"ConvNeXt encoder","expand":"convnext-1d"},{"id":"vq","label":"Normalized vector quantization","expand":"coco-vq"}],"edges":[["chroma","sum"],["whisper","sum"],["sum","down"],["down","convnext"],["convnext","vq"]]},"coco-prosody":{"title":"CoCo prosody encoding","scope":"Chromagram-only encoding captures coarse melody/prosody at 6.25 Hz. The prosody vocabulary and weights are separate from content-style tokenization.","blocks":[{"id":"chroma","label":"Chromagram frontend"},{"id":"conv","label":"Input convolution"},{"id":"down","label":"Strided convolutions + GELU"},{"id":"convnext","label":"ConvNeXt encoder","expand":"convnext-1d"},{"id":"vq","label":"Normalized vector quantization","expand":"coco-vq"}],"edges":[["chroma","conv"],["conv","down"],["down","convnext"],["convnext","vq"]]},"coco-vq":{"title":"Factorized vector quantization","scope":"A projected, L2-normalized vector is matched to normalized learned codebook entries. This is vector lookup, not finite scalar quantization.","blocks":[{"id":"project","label":"Project encoded frames"},{"id":"norm","label":"L2-normalize vectors"},{"id":"scores","label":"Compare with normalized codebook"},{"id":"ids","label":"Nearest codebook IDs","kind":"output"}],"edges":[["project","norm"],["norm","scores"],["scores","ids"]]},"diffllama":{"title":"DiffLlama acoustic flow","scope":"Target content-style tokens are expanded to acoustic frames. Reference tokens and mel frames form a prompt; conditional/unconditional velocity predictions support guided flow integration.","blocks":[{"id":"tokens","label":"Token embeddings + frame expansion"},{"id":"prompt","label":"Reference tokens / mel prompt","kind":"input"},{"id":"project","label":"Mel / token conditioning MLPs"},{"id":"time","label":"Diffusion-time embedding","kind":"input"},{"id":"layers","label":"Noncausal DiffLlama layers","expand":"diffllama-layer"},{"id":"head","label":"Adaptive RMSNorm + mel velocity MLP"},{"id":"solve","label":"Guided flow integration"}],"edges":[["tokens","project"],["prompt","project"],["project","layers"],["time","layers"],["layers","head"],["time","head"],["head","solve"]]},"diffllama-layer":{"title":"DiffLlama layer","scope":"Time-conditioned RMSNorm scales the two residual branches. Attention is bidirectional with RoPE; despite the Llama name, this is not autoregressive token decoding.","blocks":[{"id":"norm1","label":"Time-adaptive RMSNorm"},{"id":"attn","label":"RoPE self-attention"},{"id":"add1","label":"Residual add"},{"id":"norm2","label":"Time-adaptive RMSNorm"},{"id":"ffn","label":"SwiGLU feed-forward"},{"id":"add2","label":"Residual add"}],"edges":[["norm1","attn"],["attn","add1"],["add1","norm2"],["norm2","ffn"],["ffn","add2"]],"residuals":[["norm1","add1"],["norm2","add2"]]},"ctc-emissions":{"title":"Framewise CTC emissions","scope":"The alignment consumer uses full frame log-probabilities, including blanks; it does not apply greedy token collapse.","blocks":[{"id":"linear","label":"Acoustic-to-label linear projection"},{"id":"softmax","label":"Log-softmax across vocabulary"},{"id":"frames","label":"Frame / label score matrix","kind":"output"}],"edges":[["linear","softmax"],["softmax","frames"]]},"wenet-content":{"title":"WeNet content bottleneck","scope":"MeanVC2 uses chunked Fast-U2++ features. Attention and convolution retain bounded left context; the ASR text decoder is not part of conversion.","blocks":[{"id":"sub","label":"Two strided Conv2d + ReLU stages"},{"id":"proj","label":"Flatten frequency + linear projection"},{"id":"layers","label":"Relative-attention Conformer stack","expand":"conformer-block"},{"id":"norm","label":"Final LayerNorm"},{"id":"rate","label":"Bottleneck frame-rate alignment"}],"edges":[["sub","proj"],["proj","layers"],["layers","norm"],["norm","rate"]]},"wavlm":{"title":"WavLM feature encoder","scope":"The waveform CNN and positional convolution precede bidirectional attention. Feature consumers select individual layers or combine them with learned weights.","blocks":[{"id":"cnn","label":"Strided waveform CNN + GELU"},{"id":"proj","label":"Feature normalization / projection"},{"id":"pos","label":"Grouped positional convolution"},{"id":"layers","label":"Bidirectional Transformer layers","expand":"wavlm-layer"},{"id":"features","label":"Select / mix hidden layers"}],"edges":[["cnn","proj"],["proj","pos"],["pos","layers"],["layers","features"]]},"wavlm-layer":{"title":"WavLM Transformer layer","scope":"Query-dependent gates modulate bucketed relative-position bias. Pre- versus post-normalization follows the checkpoint configuration.","blocks":[{"id":"qkv","label":"Normalized Q / K / V projections"},{"id":"bias","label":"Query-gated relative-position bias"},{"id":"attn","label":"Bidirectional self-attention"},{"id":"add1","label":"Output projection + residual"},{"id":"ffn","label":"Normalized GELU feed-forward"},{"id":"add2","label":"Residual add"}],"edges":[["qkv","bias"],["qkv","attn"],["bias","attn"],["attn","add1"],["add1","ffn"],["ffn","add2"]],"residuals":[["qkv","add1"],["ffn","add2"]]},"ecapa-features":{"title":"ECAPA-TDNN speaker embedding","scope":"MeanVC2 supplies normalized WavLM features. Other ECAPA users may instead supply mel features and use different dimensions.","blocks":[{"id":"tdnn","label":"TDNN convolution + ReLU / BatchNorm"},{"id":"res","label":"SE-Res2Net temporal blocks","expand":"ecapa-res2net"},{"id":"aggregate","label":"Multi-layer feature aggregation"},{"id":"pool","label":"Attentive weighted mean / std"},{"id":"project","label":"Normalization + embedding projection"}],"edges":[["tdnn","res"],["res","aggregate"],["aggregate","pool"],["pool","project"]]},"ecapa-res2net":{"title":"SE-Res2Net block","scope":"Channel groups use hierarchical dilated temporal convolutions. Squeeze-excitation applies a learned channel gate before the residual add.","blocks":[{"id":"in","label":"Pointwise projection / activation"},{"id":"groups","label":"Split channels + hierarchical dilated convs"},{"id":"join","label":"Concatenate + pointwise projection"},{"id":"se","label":"Temporal pooling + channel gate"},{"id":"out","label":"Scale channels + residual"}],"edges":[["in","groups"],["groups","join"],["join","se"],["se","out"]],"residuals":[["in","out"]]},"meanvc-timbre":{"title":"Universal timbre-token conditioning","scope":"Speaker-to-memory MLPs produce learned key/value tokens. Content bottleneck frames query them for pronunciation-dependent timbre features.","blocks":[{"id":"speaker","label":"Speaker embedding","kind":"input"},{"id":"memory","label":"Key/value MLPs + learned priors + norm"},{"id":"content","label":"Content frame queries","kind":"input"},{"id":"attn","label":"Projected cross-attention"},{"id":"out","label":"Output projection + LayerNorm"}],"edges":[["speaker","memory"],["memory","attn"],["content","attn"],["attn","out"]]},"meanvc-dit":{"title":"MeanVC2 mean-flow generation","scope":"Noise, frame-level timbre and global speaker features feed a chunk-aware DiT. Time and reference-time embeddings modulate its blocks.","blocks":[{"id":"input","label":"Concatenate noise / timbre / speaker"},{"id":"project","label":"Input projection"},{"id":"time","label":"Time + reference-time embeddings","kind":"input"},{"id":"layers","label":"Repeated DiT blocks","expand":"meanvc-dit-layer"},{"id":"head","label":"Adaptive normalization + mel velocity"},{"id":"solve","label":"Mean-flow integration"}],"edges":[["input","project"],["project","layers"],["time","layers"],["layers","head"],["time","head"],["head","solve"]]},"meanvc-dit-layer":{"title":"Chunk-aware DiT block","scope":"Q/K RMSNorm and RoPE precede masked attention over cached/current context. Time-conditioned scale, shift and gates modulate both residual branches.","blocks":[{"id":"norm1","label":"Time-modulated LayerNorm"},{"id":"attn","label":"Q/K norm + RoPE + context attention"},{"id":"add1","label":"Time gate + residual"},{"id":"norm2","label":"Time-modulated LayerNorm"},{"id":"ffn","label":"Linear + GELU + linear"},{"id":"add2","label":"Time gate + residual"}],"edges":[["norm1","attn"],["attn","add1"],["add1","norm2"],["norm2","ffn"],["ffn","add2"]],"residuals":[["norm1","add1"],["norm2","add2"]]},"linear-spectrum":{"title":"Linear spectral frontend","scope":"This representation preserves linear-frequency bins rather than projecting them into mel bands.","blocks":[{"id":"frame","label":"Frame + window waveform"},{"id":"fft","label":"Complex FFT"},{"id":"mag","label":"Magnitude spectrum"}],"edges":[["frame","fft"],["fft","mag"]]},"tone-speaker":{"title":"Tone Color reference encoder","scope":"The final recurrent state summarizes a full source or target reference recording.","blocks":[{"id":"norm","label":"Spectrum LayerNorm"},{"id":"conv","label":"Six strided Conv2d + ReLU stages"},{"id":"reshape","label":"Flatten channel / frequency axes"},{"id":"gru","label":"GRU sequence encoder"},{"id":"out","label":"Final-state projection"}],"edges":[["norm","conv"],["conv","reshape"],["reshape","gru"],["gru","out"]]},"tone-posterior":{"title":"Posterior acoustic encoding","scope":"The source spectrum determines latent mean and scale. Sampling temperature controls Gaussian variation; no text tokens are used.","blocks":[{"id":"pre","label":"Input convolution"},{"id":"wn","label":"Gated residual / skip convolutions","expand":"wavenet-conditioned"},{"id":"stats","label":"Mean / log-scale projection"},{"id":"sample","label":"Posterior Gaussian sample"}],"edges":[["pre","wn"],["wn","stats"],["stats","sample"]]},"wavenet-conditioned":{"title":"WaveNet-style conditioning stack","scope":"Temporal convolutions and optional speaker projections feed tanh/sigmoid gates. Residual channels update the state while skip channels accumulate the output.","blocks":[{"id":"conv","label":"Temporal convolution + speaker projection"},{"id":"gate","label":"tanh branch × sigmoid branch"},{"id":"project","label":"Residual / skip projection"},{"id":"res","label":"Residual state / repeat layers"},{"id":"skip","label":"Accumulate skip outputs"}],"edges":[["conv","gate"],["gate","project"],["project","res"],["project","skip"]],"residuals":[["conv","res"]]},"tone-flow":{"title":"Additive coupling transform","scope":"Forward adds the predicted shift; inverse subtracts it and reverses the transform order. Source and target passes share weights but use different speaker conditions.","blocks":[{"id":"split","label":"Channel split / configured flip"},{"id":"net","label":"Speaker-conditioned WaveNet","expand":"wavenet-conditioned"},{"id":"shift","label":"Predict coupling shift"},{"id":"transform","label":"Add or subtract shift on other partition"},{"id":"join","label":"Rejoin / repeat coupling transforms"}],"edges":[["split","net"],["net","shift"],["shift","transform"],["split","transform"],["transform","join"]]},"mio-content":{"title":"MioCodec content encoding","scope":"WavLM content features are encoded with local rotary attention, downsampled to 25 Hz and quantized with FSQ. The quantizer is not a learned nearest-neighbor codebook.","blocks":[{"id":"encoder","label":"Local Transformer stack","expand":"mio-transformer"},{"id":"down","label":"Strided temporal convolution"},{"id":"project","label":"Project to scalar latent dimensions"},{"id":"fsq","label":"Bound + round finite scalar levels"},{"id":"out","label":"Project quantized content representation"}],"edges":[["encoder","down"],["down","project"],["project","fsq"],["fsq","out"]]},"mio-global":{"title":"MioCodec global voice encoder","scope":"Reference WavLM features pass through a ConvNeXt network and attentive statistics pooling. This branch is not a Transformer.","blocks":[{"id":"embed","label":"Conv1d + LayerNorm"},{"id":"blocks","label":"ConvNeXt blocks","expand":"convnext-1d"},{"id":"norm","label":"Final LayerNorm"},{"id":"pool","label":"Attentive weighted mean / std"},{"id":"out","label":"Projection + LayerNorm"}],"edges":[["embed","blocks"],["blocks","norm"],["norm","pool"],["pool","out"]]},"mio-transformer":{"title":"MioCodec local Transformer","scope":"Local noncausal RoPE attention and SwiGLU feed-forward layers. The waveform decoder uses global-conditioned adaptive LayerNorm/gates; the content encoder and prenet use ordinary LayerNorm.","blocks":[{"id":"norm1","label":"LayerNorm / conditioned AdaLN"},{"id":"attn","label":"Local-window RoPE self-attention"},{"id":"add1","label":"Optional gate + residual"},{"id":"norm2","label":"LayerNorm / conditioned AdaLN"},{"id":"ffn","label":"SwiGLU feed-forward"},{"id":"add2","label":"Optional gate + residual"}],"edges":[["norm1","attn"],["attn","add1"],["add1","norm2"],["norm2","ffn"],["ffn","add2"]],"residuals":[["norm1","add1"],["norm2","add2"]]},"hubert-rvc":{"title":"RVC HuBERT content encoder","scope":"Post-normalized bidirectional HuBERT features. v1 selects layer 10 plus projection; v2 selects layer 12 without the final projection.","blocks":[{"id":"cnn","label":"Strided waveform CNN / normalization / GELU"},{"id":"project","label":"Feature normalization + projection"},{"id":"pos","label":"Grouped positional convolution"},{"id":"layers","label":"Bidirectional Transformer layers","expand":"hubert-layer"},{"id":"select","label":"v1: layer 10 + projection / v2: layer 12"}],"edges":[["cnn","project"],["project","pos"],["pos","layers"],["layers","select"]]},"hubert-layer":{"title":"HuBERT post-norm layer","scope":"The RVC HuBERT checkpoint uses post-normalization. The architecture is related to Wav2Vec2, but pre-norm diagrams should not be substituted here.","blocks":[{"id":"attn","label":"Bidirectional self-attention"},{"id":"norm1","label":"Residual add + LayerNorm"},{"id":"ffn","label":"Linear + GELU + linear"},{"id":"norm2","label":"Residual add + LayerNorm"}],"edges":[["attn","norm1"],["norm1","ffn"],["ffn","norm2"]],"residuals":[["attn","norm1"],["ffn","norm2"]]},"rvc-retrieval":{"title":"Optional feature retrieval","scope":"When enabled, the index supplies similar training-frame features. When disabled, the original HuBERT content features pass through unchanged.","blocks":[{"id":"query","label":"Source feature queries"},{"id":"search","label":"Nearest-neighbor index search"},{"id":"weight","label":"Distance-weighted feature combination"},{"id":"blend","label":"Blend with original source features"}],"edges":[["query","search"],["search","weight"],["weight","blend"],["query","blend"]]},"rmvpe":{"title":"RMVPE pitch estimation","scope":"Residual encoder/decoder processing restores time-frequency resolution before recurrent pitch-bin prediction. Salience is decoded to continuous F0 with an unvoiced threshold.","blocks":[{"id":"mel","label":"Log-mel frontend"},{"id":"unet","label":"Residual U-Net with skip connections"},{"id":"gru","label":"Bidirectional GRU"},{"id":"head","label":"Pitch-bin projection + sigmoid"},{"id":"f0","label":"Local salience decoding / voicing threshold"}],"edges":[["mel","unet"],["unet","gru"],["gru","head"],["head","f0"]]},"rvc-prior":{"title":"RVC acoustic latent prior","scope":"HuBERT features replace a text-token embedding frontend. F0 checkpoints add a coarse pitch embedding before relative attention.","blocks":[{"id":"content","label":"Content projection + optional pitch embedding"},{"id":"layers","label":"Relative-attention encoder","expand":"vits-text-layer"},{"id":"stats","label":"Latent mean / log-scale projection"},{"id":"sample","label":"Gaussian prior sampling"}],"edges":[["content","layers"],["layers","stats"],["stats","sample"]]},"rvc-nsf":{"title":"F0-conditioned waveform synthesis","scope":"Continuous F0 produces harmonic/noise excitation. Each upsampling stage combines it with speaker-conditioned latent features before residual synthesis.","blocks":[{"id":"latent","label":"Latent convolution + speaker condition"},{"id":"f0","label":"Continuous F0","kind":"input"},{"id":"source","label":"Harmonic / noise source + learned mixing"},{"id":"up","label":"Transposed convolution + excitation injection"},{"id":"res","label":"Multi-receptive-field residual bank"},{"id":"repeat","label":"Repeat upsampling stages"},{"id":"out","label":"Output convolution + tanh"}],"edges":[["latent","up"],["f0","source"],["source","up"],["up","res"],["res","repeat"],["repeat","out"]]},"sam-t5":{"title":"T5 text conditioning","scope":"SAM Audio uses the ReLU T5 encoder, not a causal language-model decoder.","blocks":[{"id":"tokens","label":"SentencePiece + EOS"},{"id":"embed","label":"Token embeddings"},{"id":"layers","label":"Bidirectional T5 layers","expand":"sam-t5-layer"},{"id":"norm","label":"Final RMSNorm"}],"edges":[["tokens","embed"],["embed","layers"],["layers","norm"]]},"sam-t5-layer":{"title":"T5 encoder layer","scope":"Learned bucketed relative-position bias is added to self-attention scores. Feed-forward layers use ReLU.","blocks":[{"id":"norm1","label":"RMSNorm"},{"id":"attn","label":"Self-attention + relative bias"},{"id":"add1","label":"Residual add"},{"id":"norm2","label":"RMSNorm"},{"id":"ffn","label":"Linear + ReLU + linear"},{"id":"add2","label":"Residual add"}],"edges":[["norm1","attn"],["attn","add1"],["add1","norm2"],["norm2","ffn"],["ffn","add2"]],"residuals":[["norm1","add1"],["norm2","add2"]]},"sam-vision":{"title":"Perception Encoder visual conditioning","scope":"Frames use 14-pixel patches and spatial rotary attention. Learned-query pooling yields frame features; video features are aligned to audio latent frames.","blocks":[{"id":"patch","label":"Patch convolution + class token"},{"id":"pos","label":"Position embeddings + LayerNorm"},{"id":"layers","label":"Vision Transformer layers","expand":"sam-vision-layer"},{"id":"norm","label":"Final LayerNorm"},{"id":"pool","label":"Learned-query attention pooling + MLP"},{"id":"proj","label":"Projection + L2 normalization"},{"id":"align","label":"Align features to audio frames"}],"edges":[["patch","pos"],["pos","layers"],["layers","norm"],["norm","pool"],["pool","proj"],["proj","align"]]},"sam-vision-layer":{"title":"Perception Encoder ViT layer","scope":"Pre-normalized noncausal attention uses horizontal and vertical RoPE on query/key channels. A GELU MLP follows.","blocks":[{"id":"norm1","label":"LayerNorm"},{"id":"attn","label":"QKV + spatial RoPE + attention"},{"id":"add1","label":"Projection + residual add"},{"id":"norm2","label":"LayerNorm"},{"id":"ffn","label":"Linear + GELU + linear"},{"id":"add2","label":"Residual add"}],"edges":[["norm1","attn"],["attn","add1"],["add1","norm2"],["norm2","ffn"],["ffn","add2"]],"residuals":[["norm1","add1"],["norm2","add2"]]},"sam-anchors":{"title":"Temporal prompt conditioning","scope":"Each latent frame receives a positive, negative or unspecified label. This is learned conditioning, not cropping the recording.","blocks":[{"id":"spans","label":"Map time spans to latent frames"},{"id":"lookup","label":"Anchor label embedding"},{"id":"project","label":"Linear projection"},{"id":"gate","label":"Learned tanh gate"}],"edges":[["spans","lookup"],["lookup","project"],["project","gate"]]},"sam-dit":{"title":"SAM Audio flow-matching DiT","scope":"The same mixture conditions target and residual generation. The sampler uses two velocity evaluations per midpoint step, not autoregressive token generation.","blocks":[{"id":"mix","label":"Mixture + noisy separated latents","kind":"input"},{"id":"project","label":"Concatenate + input projection"},{"id":"condition","label":"Add gated visual / anchor features"},{"id":"conv","label":"Residual temporal convolution"},{"id":"time","label":"Time embedding","kind":"input"},{"id":"text","label":"Projected T5 memory","kind":"input"},{"id":"layers","label":"Repeated conditioned DiT layers","expand":"sam-dit-layer"},{"id":"head","label":"Time-modulated norm + velocity projection"},{"id":"step","label":"Midpoint ODE update / repeat steps"}],"edges":[["mix","project"],["project","condition"],["condition","conv"],["conv","layers"],["time","layers"],["text","layers"],["layers","head"],["time","head"],["head","step"]]},"sam-dit-layer":{"title":"SAM Audio DiT layer","scope":"Time conditioning modulates self-attention and SwiGLU branches. Text cross-attention sits between them; self-attention uses RoPE with configured Q/K normalization.","blocks":[{"id":"norm1","label":"RMSNorm + time scale / shift"},{"id":"attn","label":"RoPE self-attention"},{"id":"add1","label":"Time gate + residual add"},{"id":"cross","label":"Text cross-attention"},{"id":"add2","label":"Residual add"},{"id":"norm2","label":"RMSNorm + time scale / shift"},{"id":"ffn","label":"SwiGLU feed-forward"},{"id":"add3","label":"Time gate + residual add"}],"edges":[["norm1","attn"],["attn","add1"],["add1","cross"],["cross","add2"],["add2","norm2"],["norm2","ffn"],["ffn","add3"]],"residuals":[["norm1","add1"],["cross","add2"],["norm2","add3"]]},"sam-dac-residual":{"title":"DAC Snake residual unit","scope":"Three units per stage use dilation 1, 3 and 9. Snake is a learned periodic activation.","blocks":[{"id":"snake1","label":"Snake activation"},{"id":"conv","label":"Dilated Conv1d"},{"id":"snake2","label":"Snake activation"},{"id":"point","label":"Pointwise convolution"},{"id":"add","label":"Residual add"}],"edges":[["snake1","conv"],["conv","snake2"],["snake2","point"],["point","add"]],"residuals":[["snake1","add"]]},"sam-dac-encoder":{"title":"Continuous DAC-VAE encoding","scope":"The inference encoder selects latent mean channels from its projection. There is no residual vector quantization or token sampling here.","blocks":[{"id":"conv","label":"Input convolution"},{"id":"res","label":"Snake residual units","expand":"sam-dac-residual"},{"id":"down","label":"Snake + strided convolution"},{"id":"repeat","label":"Repeat downsampling stages"},{"id":"project","label":"Final convolution + latent projection"},{"id":"mean","label":"Select continuous latent mean"}],"edges":[["conv","res"],["res","down"],["down","repeat"],["repeat","project"],["project","mean"]]},"sam-dac-decoder":{"title":"DAC-VAE waveform synthesis","scope":"Target and residual streams share decoder weights. This diagram shows the active unwatermarked audio.cpp output path, not the retained watermark subnetwork.","blocks":[{"id":"project","label":"Latent projection + input convolution"},{"id":"up","label":"Snake + transposed convolution"},{"id":"res","label":"Snake residual units","expand":"sam-dac-residual"},{"id":"repeat","label":"Repeat upsampling stages"},{"id":"out","label":"Snake + output convolution + tanh"}],"edges":[["project","up"],["up","res"],["res","repeat"],["repeat","out"]]},"sano-student":{"title":"SanoTTS acoustic student","scope":"Both lineages use residual convolutional stages. Nano predicts mel frames; PiperLite predicts acoustic decoder latents.","blocks":[{"id":"tokens","label":"Phoneme embeddings + positional features"},{"id":"duration","label":"Residual conv duration network"},{"id":"context","label":"Residual conv token-context network"},{"id":"expand","label":"Duration-based context expansion"},{"id":"frames","label":"Frame features + residual conv network"},{"id":"output","label":"Mel / latent projection"}],"edges":[["tokens","duration"],["tokens","context"],["duration","context"],["duration","expand"],["context","expand"],["expand","frames"],["frames","output"]]},"sano-nano-decoder":{"title":"Nano spectral decoder","scope":"Noise is injected into mel conditioning before the ConvNeXt stack. Log magnitude and phase are decoded with inverse STFT.","blocks":[{"id":"mel","label":"Mel projection"},{"id":"noise","label":"Noise adapter"},{"id":"sum","label":"Add + LayerNorm"},{"id":"blocks","label":"ConvNeXt residual blocks","expand":"convnext-1d"},{"id":"head","label":"Log-magnitude / phase projection"},{"id":"istft","label":"Inverse STFT","expand":"inverse-stft"}],"edges":[["mel","sum"],["noise","sum"],["sum","blocks"],["blocks","head"],["head","istft"]]},"sano-piper-decoder":{"title":"PiperLite waveform decoder","scope":"A checkpoint-specific latent adapter may precede the decoder, and a learned postfilter may follow it. No VITS coupling stack is used.","blocks":[{"id":"adapter","label":"Latent adaptation / input convolution"},{"id":"up","label":"Three transposed-convolution stages"},{"id":"res","label":"Residual convolution banks"},{"id":"out","label":"Output convolution + tanh"},{"id":"filter","label":"Optional learned postfilter"}],"edges":[["adapter","up"],["up","res"],["res","out"],["out","filter"]]},"convnext-1d":{"title":"Temporal ConvNeXt block","scope":"Temporal depthwise convolution and a channel-wise MLP. Dilation, causal padding and residual scaling depend on the owning component.","blocks":[{"id":"depth","label":"Depthwise Conv1d"},{"id":"norm","label":"LayerNorm"},{"id":"up","label":"Channel expansion + GELU"},{"id":"down","label":"Channel projection / LayerScale"},{"id":"sum","label":"Residual add"}],"edges":[["depth","norm"],["norm","up"],["up","down"],["down","sum"]],"residuals":[["depth","sum"]]},"supertonic-tokenizer":{"title":"Supertonic text preparation","scope":"The model reads character IDs, including language/expression formatting; no grapheme-to-phoneme stage is used.","blocks":[{"id":"format","label":"Normalize text / language formatting"},{"id":"unicode","label":"Unicode NFKD normalization"},{"id":"lookup","label":"Character indexer lookup"}],"edges":[["format","unicode"],["unicode","lookup"]]},"supertonic-duration":{"title":"Supertonic duration network","scope":"Sentence-level duration sets latent length. Stored duration-style conditioning is separate from text-to-latent style conditioning.","blocks":[{"id":"embed","label":"Character embedding + sentence token"},{"id":"conv","label":"ConvNeXt stack","expand":"convnext-1d"},{"id":"attn","label":"Self-attention encoder"},{"id":"pool","label":"Select sentence-token representation"},{"id":"style","label":"Duration style","kind":"input"},{"id":"mlp","label":"Concatenate + MLP + exponential"}],"edges":[["embed","conv"],["conv","attn"],["attn","pool"],["pool","mlp"],["style","mlp"]]},"supertonic-text":{"title":"Supertonic text conditioning","scope":"Stored style values and learned style keys provide voice conditioning through two cross-attention stages.","blocks":[{"id":"embed","label":"Character embedding"},{"id":"conv","label":"Dilated ConvNeXt stack","expand":"convnext-1d"},{"id":"attn","label":"Self-attention encoder + residual"},{"id":"style","label":"Stored style values / learned keys","kind":"input"},{"id":"cross","label":"Style cross-attention stages"},{"id":"norm","label":"LayerNorm"}],"edges":[["embed","conv"],["conv","attn"],["attn","cross"],["style","cross"],["cross","norm"]]},"supertonic-flow":{"title":"Supertonic latent flow","scope":"Each integration step predicts conditional and unconditional vector fields and combines them with classifier-free guidance.","blocks":[{"id":"project","label":"Noisy latent projection"},{"id":"conv","label":"Dilated ConvNeXt blocks","expand":"convnext-1d"},{"id":"time","label":"Add time conditioning"},{"id":"text","label":"ConvNeXt + text-memory attention"},{"id":"style","label":"ConvNeXt + style-memory attention"},{"id":"repeat","label":"Repeat block groups / final ConvNeXt"},{"id":"head","label":"Vector-field projection + CFG"},{"id":"step","label":"Euler latent update"}],"edges":[["project","conv"],["conv","time"],["time","text"],["text","style"],["style","repeat"],["repeat","head"],["head","step"]]},"supertonic-decoder":{"title":"Supertonic waveform decoder","scope":"Latent denormalization and temporal unpacking precede causal convolutional synthesis. The final projection emits sample patches, not spectral coefficients.","blocks":[{"id":"unpack","label":"Unpack / denormalize latent frames"},{"id":"embed","label":"Causal input convolution"},{"id":"conv","label":"Causal dilated ConvNeXt stack","expand":"convnext-1d"},{"id":"norm","label":"BatchNorm"},{"id":"head","label":"Convolution + PReLU + sample projection"},{"id":"wave","label":"Concatenate sample patches"}],"edges":[["unpack","embed"],["embed","conv"],["conv","norm"],["norm","head"],["head","wave"]]},"phoneme-frontend":{"title":"Phoneme text frontend","scope":"Normalization, phoneme conventions and symbol mapping follow the model and language. Sharing this pipeline does not make vocabularies interchangeable.","blocks":[{"id":"normalize","label":"Text normalization"},{"id":"g2p","label":"Language-specific G2P"},{"id":"post","label":"Phoneme postprocessing"},{"id":"tokens","label":"Vocabulary IDs + boundary symbols"}],"edges":[["normalize","g2p"],["g2p","post"],["post","tokens"]]},"preset-style":{"title":"Stored voice conditioning","scope":"Kokoro and Kitten packages contain length-indexed style rows. Acoustic and prosodic style portions condition different parts of synthesis.","blocks":[{"id":"voice","label":"Voice selection","kind":"input"},{"id":"length","label":"Phoneme count","kind":"input"},{"id":"lookup","label":"Select stored style row"},{"id":"split","label":"Acoustic / prosodic style split"}],"edges":[["voice","lookup"],["length","lookup"],["lookup","split"]]},"plbert":{"title":"PL-BERT phoneme encoder","scope":"ALBERT factorizes the embedding width and shares Transformer layer parameters across depth. This is bidirectional phoneme encoding, not AR text generation.","blocks":[{"id":"embed","label":"Phoneme + position + type embeddings"},{"id":"norm","label":"LayerNorm + hidden projection"},{"id":"layers","label":"Repeated shared ALBERT layer","expand":"albert-layer"},{"id":"out","label":"Prosody-width projection"}],"edges":[["embed","norm"],["norm","layers"],["layers","out"]]},"albert-layer":{"title":"ALBERT layer","scope":"Post-normalized bidirectional self-attention and GELU feed-forward network.","blocks":[{"id":"attn","label":"Multi-head self-attention"},{"id":"norm","label":"Residual add + LayerNorm"},{"id":"mlp","label":"Linear + GELU + linear"},{"id":"out","label":"Residual add + LayerNorm"}],"edges":[["attn","norm"],["norm","mlp"],["mlp","out"]],"residuals":[["attn","norm"],["mlp","out"]]},"styletts-text":{"title":"StyleTTS-derived text encoder","scope":"This branch reads phoneme IDs independently of PL-BERT. Its features are later expanded to acoustic frames by predicted durations.","blocks":[{"id":"embed","label":"Phoneme embeddings"},{"id":"conv","label":"Convolution / normalization / activation stack"},{"id":"lstm","label":"Bidirectional LSTM"},{"id":"out","label":"Token-level synthesis features","kind":"output"}],"edges":[["embed","conv"],["conv","lstm"],["lstm","out"]]},"styletts-prosody":{"title":"Style-conditioned duration and prosody","scope":"Duration expansion aligns both contextual prosody features and the independent text-encoder features. The resulting frame features and pitch/noise predictions condition synthesis.","blocks":[{"id":"bert","label":"PL-BERT features","kind":"input"},{"id":"style","label":"Prosodic style","kind":"input"},{"id":"encoder","label":"Style-conditioned recurrent text stack"},{"id":"duration","label":"BiLSTM + duration projection"},{"id":"align","label":"Duration-based frame expansion"},{"id":"text","label":"Text-encoder features","kind":"input"},{"id":"shared","label":"Shared recurrent prosody layer"},{"id":"f0","label":"AdaIN residual pitch head"},{"id":"noise","label":"AdaIN residual noise head"}],"edges":[["bert","encoder"],["style","encoder"],["encoder","duration"],["duration","align"],["encoder","align"],["text","align"],["align","shared"],["shared","f0"],["shared","noise"],["style","f0"],["style","noise"]]},"styletts-istft":{"title":"Style-conditioned inverse-STFT synthesis","scope":"Pitch drives harmonic excitation. Residual upsampling uses stored acoustic style, not a sampled diffusion style.","blocks":[{"id":"features","label":"Aligned text + pitch / noise features","kind":"input"},{"id":"style","label":"Acoustic style","kind":"input"},{"id":"acoustic","label":"AdaIN residual acoustic decoder"},{"id":"source","label":"Pitch-driven harmonics + noise"},{"id":"spectrum","label":"Excitation STFT"},{"id":"up","label":"Upsampling / adaptive residual blocks"},{"id":"head","label":"Magnitude / phase projection"},{"id":"istft","label":"Inverse STFT","expand":"inverse-stft"}],"edges":[["features","acoustic"],["style","acoustic"],["features","source"],["source","spectrum"],["spectrum","up"],["acoustic","up"],["style","up"],["up","head"],["head","istft"]]},"vits-text":{"title":"VITS phoneme encoder","scope":"Bidirectional relative attention and convolutional feed-forward layers predict a Gaussian latent prior for each phoneme.","blocks":[{"id":"embed","label":"Phoneme embeddings"},{"id":"layers","label":"Relative-attention encoder layers","expand":"vits-text-layer"},{"id":"stats","label":"Prior mean / log-scale projection"}],"edges":[["embed","layers"],["layers","stats"]]},"vits-text-layer":{"title":"VITS text Transformer layer","scope":"Relative-position self-attention is followed by a convolutional feed-forward branch. Both use post-residual normalization.","blocks":[{"id":"attn","label":"Relative self-attention"},{"id":"norm","label":"Residual add + LayerNorm"},{"id":"conv","label":"Conv1d + ReLU + Conv1d"},{"id":"out","label":"Residual add + LayerNorm"}],"edges":[["attn","norm"],["norm","conv"],["conv","out"]],"residuals":[["attn","norm"],["conv","out"]]},"vits-stochastic-duration":{"title":"Stochastic duration inference","scope":"Duration sampling is separate from the acoustic latent coupling flow. Piper uses text-conditioned inverse rational-quadratic spline transforms.","blocks":[{"id":"text","label":"Text-encoder hidden features","kind":"input"},{"id":"condition","label":"Pointwise + dilated depthwise convolutions"},{"id":"noise","label":"Duration noise","kind":"input"},{"id":"flow","label":"Inverse conditioned spline flows"},{"id":"duration","label":"Log duration / exponentiate / speed scaling"}],"edges":[["text","condition"],["condition","flow"],["noise","flow"],["flow","duration"]]},"vits-duration":{"title":"Deterministic duration predictor","scope":"Inflect predicts log-duration directly from text features; there is no stochastic duration-flow stack.","blocks":[{"id":"conv1","label":"Conv1d + ReLU + LayerNorm"},{"id":"conv2","label":"Conv1d + ReLU + LayerNorm"},{"id":"out","label":"Scalar log-duration projection"}],"edges":[["conv1","conv2"],["conv2","out"]]},"vits-alignment":{"title":"VITS prior sampling","scope":"Duration controls the number of acoustic frames assigned to each phoneme. Acoustic noise variation remains independent of duration prediction.","blocks":[{"id":"duration","label":"Scaled durations","kind":"input"},{"id":"prior","label":"Token prior mean / log-scale","kind":"input"},{"id":"expand","label":"Repeat prior statistics by duration"},{"id":"sample","label":"Mean + scaled Gaussian noise"}],"edges":[["duration","expand"],["prior","expand"],["expand","sample"]]},"vits-coupling":{"title":"VITS inverse coupling stack","scope":"Repeated channel flips and residual coupling transforms invert the learned prior mapping. The conditioning network uses gated dilated convolutions.","blocks":[{"id":"flip","label":"Channel flip / split"},{"id":"condition","label":"WaveNet-style gated dilated convolutions"},{"id":"couple","label":"Inverse residual coupling"},{"id":"join","label":"Join channels / repeat flows"}],"edges":[["flip","condition"],["condition","couple"],["flip","couple"],["couple","join"]]},"vits-waveform":{"title":"VITS waveform decoder","scope":"HiFi-GAN-style generator without a separate mel prediction stage. Continuous latents directly condition the upsampling network.","blocks":[{"id":"pre","label":"Input convolution"},{"id":"up","label":"LeakyReLU + transposed convolution"},{"id":"res","label":"Multi-receptive-field residual convolutions"},{"id":"repeat","label":"Repeat upsampling stages"},{"id":"out","label":"Output convolution + tanh"}],"edges":[["pre","up"],["up","res"],["res","repeat"],["repeat","out"]]},"demucs-encoder":{"title":"Demucs convolutional encoder stage","scope":"Waveform stages stride along time; spectral stages stride along frequency. Normalization and rewrite blocks follow the checkpoint stage configuration.","blocks":[{"id":"conv","label":"Strided convolution"},{"id":"act","label":"Normalization / GELU"},{"id":"res","label":"Residual dilated temporal blocks","expand":"demucs-dconv"},{"id":"rewrite","label":"Convolutional rewrite + GLU"},{"id":"skip","label":"Retain encoder skip features"}],"edges":[["conv","act"],["act","res"],["res","rewrite"],["rewrite","skip"]]},"demucs-dconv":{"title":"Demucs residual temporal block","scope":"A dilated bottleneck convolution is expanded through a GLU and scaled before residual addition.","blocks":[{"id":"conv","label":"Dilated Conv1d"},{"id":"norm","label":"GroupNorm + GELU"},{"id":"project","label":"Pointwise convolution"},{"id":"glu","label":"GroupNorm + GLU"},{"id":"scale","label":"LayerScale"},{"id":"sum","label":"Residual add"}],"edges":[["conv","norm"],["norm","project"],["project","glu"],["glu","scale"],["scale","sum"]],"residuals":[["conv","sum"]]},"demucs-transformer":{"title":"Hybrid Transformer bottleneck","scope":"Both cross-attention directions read the previous pair of states, not a sequentially updated partner. Self- and cross-attention stages alternate.","blocks":[{"id":"wave","label":"Waveform features + positions","kind":"input"},{"id":"spec","label":"Spectral features + positions","kind":"input"},{"id":"waveself","label":"Waveform self-attention layer","expand":"demucs-self-layer"},{"id":"specself","label":"Spectral self-attention layer","expand":"demucs-self-layer"},{"id":"wavecross","label":"Waveform queries / spectral memory","expand":"demucs-cross-layer"},{"id":"speccross","label":"Spectral queries / waveform memory","expand":"demucs-cross-layer"}],"edges":[["wave","waveself"],["spec","specself"],["waveself","wavecross"],["specself","wavecross"],["specself","speccross"],["waveself","speccross"]]},"demucs-self-layer":{"title":"Demucs self-attention layer","scope":"Pre-normalized non-causal attention and feed-forward branches use learned residual scales. Output normalization is checkpoint-controlled.","blocks":[{"id":"norm","label":"LayerNorm"},{"id":"attn","label":"Self-attention + LayerScale"},{"id":"sum","label":"Residual add"},{"id":"ff","label":"LayerNorm + feed-forward + LayerScale"},{"id":"out","label":"Residual add / output normalization"}],"edges":[["norm","attn"],["attn","sum"],["sum","ff"],["ff","out"]],"residuals":[["norm","sum"],["ff","out"]]},"demucs-cross-layer":{"title":"Demucs cross-attention layer","scope":"Query and memory come from different audio domains and are separately normalized.","blocks":[{"id":"query","label":"Query-stream LayerNorm"},{"id":"memory","label":"Other-stream LayerNorm"},{"id":"attn","label":"Cross-attention + LayerScale"},{"id":"sum","label":"Query residual add"},{"id":"ff","label":"LayerNorm + feed-forward + LayerScale"},{"id":"out","label":"Residual add / output normalization"}],"edges":[["query","attn"],["memory","attn"],["attn","sum"],["sum","ff"],["ff","out"]],"residuals":[["query","sum"],["ff","out"]]},"demucs-decoder":{"title":"Demucs decoder stage","scope":"The waveform and spectral branches restore their respective time and frequency resolutions. Final projections produce source channels.","blocks":[{"id":"skip","label":"Add encoder skip"},{"id":"rewrite","label":"Convolutional rewrite + GLU"},{"id":"res","label":"Residual temporal blocks","expand":"demucs-dconv"},{"id":"up","label":"Transposed convolution"},{"id":"out","label":"Normalize / activate / trim"}],"edges":[["skip","rewrite"],["rewrite","res"],["res","up"],["up","out"]]},"demucs-sum":{"title":"Hybrid source reconstruction","scope":"Each source has a waveform prediction and an inverse-STFT prediction. These are added, not concatenated as separate stems.","blocks":[{"id":"wave","label":"Waveform prediction","kind":"input"},{"id":"spec","label":"Inverse-STFT prediction","kind":"input"},{"id":"sum","label":"Undo normalization + sum branches"},{"id":"out","label":"Source waveform","kind":"output"}],"edges":[["wave","sum"],["spec","sum"],["sum","out"]]},"inverse-stft":{"title":"Inverse spectral synthesis","scope":"Window, hop and length handling follow the analysis frontend.","blocks":[{"id":"fft","label":"Inverse Fourier transform"},{"id":"window","label":"Synthesis window"},{"id":"overlap","label":"Overlap-add + window normalization"},{"id":"crop","label":"Restore waveform length"}],"edges":[["fft","window"],["window","overlap"],["overlap","crop"]]},"apollo-bands":{"title":"Apollo band projection","scope":"Each band includes normalized real and imaginary bins plus a band-level energy feature.","blocks":[{"id":"split","label":"Partition complex spectrum"},{"id":"normalize","label":"Band energy normalization"},{"id":"features","label":"Complex bins + energy feature"},{"id":"norm","label":"Per-band RMSNorm"},{"id":"linear","label":"Per-band linear projection"}],"edges":[["split","normalize"],["normalize","features"],["features","norm"],["norm","linear"]]},"apollo-block":{"title":"Apollo restoration block","scope":"Attention runs across bands. Temporal convolutions run independently per band; there is no autoregressive token loop.","blocks":[{"id":"norm","label":"RMSNorm"},{"id":"attention","label":"Across-band RoPE attention"},{"id":"add","label":"Projection + residual add"},{"id":"ff","label":"RMSNorm + SiLU / gated MLP + residual"},{"id":"time","label":"Three temporal residual blocks","expand":"apollo-tcn"}],"edges":[["norm","attention"],["attention","add"],["add","ff"],["ff","time"]],"residuals":[["norm","add"]]},"apollo-tcn":{"title":"Apollo temporal block","scope":"Depthwise temporal convolution followed by a pointwise channel MLP.","blocks":[{"id":"conv","label":"Depthwise temporal convolution"},{"id":"norm","label":"RMSNorm"},{"id":"expand","label":"Linear expansion + SiLU"},{"id":"project","label":"Linear projection"},{"id":"sum","label":"Residual add"}],"edges":[["conv","norm"],["norm","expand"],["expand","project"],["project","sum"]],"residuals":[["conv","sum"]]},"apollo-head":{"title":"Apollo spectral head","scope":"The output is a directly predicted complex spectrum, not a complex multiplicative mask.","blocks":[{"id":"norm","label":"Per-band RMSNorm"},{"id":"linear","label":"Per-band linear projection"},{"id":"glu","label":"GLU"},{"id":"join","label":"Concatenate real / imaginary bands"}],"edges":[["norm","linear"],["linear","glu"],["glu","join"]]},"universr-analysis":{"title":"UniverSR analysis","scope":"Input bandwidth determines which bins are observed. Spectral magnitude is compressed while phase is retained.","blocks":[{"id":"resample","label":"Bandwidth preparation / resampling"},{"id":"stft","label":"Complex STFT","expand":"complex-stft"},{"id":"compress","label":"Magnitude compression with phase"},{"id":"low","label":"Observed low-band bins"}],"edges":[["resample","stft"],["stft","compress"],["compress","low"]]},"universr-conditioning":{"title":"UniverSR low-band encoder","scope":"Frequency and sample-rate embeddings modulate low-band features. Frequency pooling preserves the temporal sequence.","blocks":[{"id":"freq","label":"Frequency-position FiLM"},{"id":"project","label":"Complex-channel projection"},{"id":"rate","label":"Sample-rate FiLM"},{"id":"blocks","label":"ConvNeXt blocks","expand":"convnext-grn"},{"id":"pool","label":"Frequency mean pooling"},{"id":"spatial","label":"High-band frequency modulation"}],"edges":[["freq","project"],["project","rate"],["rate","blocks"],["blocks","pool"],["pool","spatial"]]},"convnext-grn":{"title":"ConvNeXt with GRN","scope":"Spatial depthwise convolution and a channel MLP. Global response normalization measures spatial energy and normalizes it across channels.","blocks":[{"id":"conv","label":"7x7 depthwise convolution"},{"id":"norm","label":"Channel LayerNorm"},{"id":"expand","label":"Linear expansion + GELU"},{"id":"grn","label":"Global response normalization"},{"id":"project","label":"Linear projection"},{"id":"sum","label":"Residual add"}],"edges":[["conv","norm"],["norm","expand"],["expand","grn"],["grn","project"],["project","sum"]],"residuals":[["conv","sum"]]},"universr-flow":{"title":"UniverSR flow U-Net","scope":"The U-Net is evaluated repeatedly along the flow trajectory. Time and sample-rate embeddings condition the ConvNeXt blocks; audio-derived conditioning enters the input projection.","blocks":[{"id":"input","label":"Noisy spectrum + conditioning projection"},{"id":"down","label":"Four ConvNeXt / downsampling stages","expand":"universr-time-block"},{"id":"mid","label":"Bottleneck ConvNeXt blocks","expand":"universr-time-block"},{"id":"up","label":"Four upsampling / ConvNeXt stages","expand":"universr-time-block"},{"id":"out","label":"Complex vector-field projection"},{"id":"step","label":"Flow integration step"}],"edges":[["input","down"],["down","mid"],["mid","up"],["down","up"],["up","out"],["out","step"]]},"universr-time-block":{"title":"Time-conditioned ConvNeXt block","scope":"Projected time/sample-rate conditioning is added to the hidden features before the ConvNeXt block.","blocks":[{"id":"time","label":"Time / sample-rate embedding","kind":"input"},{"id":"mlp","label":"Linear + SiLU + linear"},{"id":"features","label":"Hidden features","kind":"input"},{"id":"add","label":"Add conditioning"},{"id":"block","label":"ConvNeXt + GRN","expand":"convnext-grn"}],"edges":[["time","mlp"],["mlp","add"],["features","add"],["add","block"]]},"universr-synthesis":{"title":"UniverSR reconstruction","scope":"Observed low-frequency bins take precedence over the generated high-band region.","blocks":[{"id":"low","label":"Observed low band","kind":"input"},{"id":"high","label":"Generated high band","kind":"input"},{"id":"join","label":"Assemble spectrum"},{"id":"decompress","label":"Undo magnitude compression"},{"id":"istft","label":"Inverse STFT","expand":"inverse-stft"}],"edges":[["low","join"],["high","join"],["join","decompress"],["decompress","istft"]]},"roformer-bands":{"title":"Band embedding","scope":"BS-RoFormer uses disjoint bands; Mel-Band RoFormer uses overlapping mel-spaced selections.","blocks":[{"id":"split","label":"Select frequency bins per band"},{"id":"flatten","label":"Pack complex channels"},{"id":"norm","label":"Per-band RMSNorm"},{"id":"project","label":"Per-band linear projection"}],"edges":[["split","flatten"],["flatten","norm"],["norm","project"]]},"roformer-axial":{"title":"Axial RoFormer stack","scope":"Each stage alternates time and band axes. Attention is bidirectional, not autoregressive.","blocks":[{"id":"time","label":"Time-axis Transformer","expand":"roformer-layer"},{"id":"freq","label":"Band-axis Transformer","expand":"roformer-layer"},{"id":"repeat","label":"Repeat axial stages"},{"id":"norm","label":"Final normalization"}],"edges":[["time","freq"],["freq","repeat"],["repeat","norm"]]},"roformer-layer":{"title":"RoFormer Transformer layer","scope":"RMS-normalized rotary attention includes learned sigmoid gates per attention head. A GELU feed-forward branch follows.","blocks":[{"id":"norm","label":"RMSNorm"},{"id":"qkv","label":"Q / K / V projection + RoPE"},{"id":"attn","label":"Non-causal attention"},{"id":"gate","label":"Per-head sigmoid gate + projection"},{"id":"add","label":"Residual add"},{"id":"norm2","label":"RMSNorm"},{"id":"mlp","label":"Linear + GELU + linear"},{"id":"sum","label":"Residual add"}],"edges":[["norm","qkv"],["qkv","attn"],["attn","gate"],["gate","add"],["add","norm2"],["norm2","mlp"],["mlp","sum"]],"residuals":[["norm","add"],["norm2","sum"]]},"roformer-mask":{"title":"RoFormer mask estimator","scope":"Band-specific MLPs emit real and imaginary mask values. Overlap averaging applies to Mel-Band RoFormer; disjoint bands simply assemble.","blocks":[{"id":"mlp","label":"Band-specific MLP"},{"id":"glu","label":"GLU complex mask output"},{"id":"assemble","label":"Assemble / average overlapping bins"}],"edges":[["mlp","glu"],["glu","assemble"]]},"complex-stft":{"title":"Complex spectral analysis","scope":"Window and hop sizes are checkpoint-specific. Real and imaginary channels preserve phase information.","blocks":[{"id":"window","label":"Framing + window"},{"id":"fft","label":"Real FFT"},{"id":"complex","label":"Complex spectrum","kind":"output"}],"edges":[["window","fft"],["fft","complex"]]},"rnnoise-analysis":{"title":"RNNoise feature analysis","scope":"Signal-processing features, not a learned mel encoder. Spectrum and pitch state also feed synthesis.","blocks":[{"id":"audio","label":"Audio frame","kind":"input"},{"id":"fft","label":"Windowed Fourier analysis","expand":"complex-stft"},{"id":"bands","label":"Band energies / cepstra"},{"id":"pitch","label":"Pitch search + correlations"},{"id":"features","label":"Concatenate features"}],"edges":[["audio","fft"],["fft","bands"],["audio","pitch"],["bands","features"],["pitch","features"]]},"rnnoise-network":{"title":"RNNoise recurrent predictor","scope":"Topology of the packaged modern RNNoise checkpoint. Hidden states persist across frames.","blocks":[{"id":"conv","label":"Two causal convolutions"},{"id":"g1","label":"GRU 1"},{"id":"g2","label":"GRU 2"},{"id":"g3","label":"GRU 3"},{"id":"concat","label":"Concatenate convolution + GRU states"},{"id":"gain","label":"Linear + sigmoid band gains"},{"id":"vad","label":"Linear + sigmoid speech score"}],"edges":[["conv","g1"],["g1","g2"],["g2","g3"],["conv","concat"],["g1","concat"],["g2","concat"],["g3","concat"],["concat","gain"],["concat","vad"]]},"rnnoise-synthesis":{"title":"RNNoise reconstruction","scope":"Predicted band gains act on the analyzed signal; no neural waveform decoder is used.","blocks":[{"id":"pitch","label":"Pitch filter"},{"id":"smooth","label":"Temporal gain smoothing"},{"id":"interp","label":"Interpolate band gains to FFT bins"},{"id":"multiply","label":"Apply spectral gains"},{"id":"synthesis","label":"Inverse FFT + overlap-add"}],"edges":[["pitch","multiply"],["smooth","interp"],["interp","multiply"],["multiply","synthesis"]]},"deepfilter-analysis":{"title":"DeepFilterNet2 analysis","scope":"Two normalized representations are extracted from one STFT. The full complex spectrum is retained for reconstruction.","blocks":[{"id":"stft","label":"STFT","expand":"complex-stft"},{"id":"erb","label":"ERB energies + normalization"},{"id":"complex","label":"Low-frequency complex bins + normalization"}],"edges":[["stft","erb"],["stft","complex"]]},"deepfilter-network":{"title":"DeepFilterNet2 network","scope":"Convolutional skip features feed the two decoders. Grouped recurrent processing links successive frames.","blocks":[{"id":"erb","label":"ERB convolution stack"},{"id":"spec","label":"Complex-feature convolutions"},{"id":"project","label":"Grouped linear projection"},{"id":"join","label":"Concatenate features"},{"id":"gru","label":"Squeezed grouped GRU"},{"id":"mask","label":"GRU + convolutional gain decoder"},{"id":"coef","label":"GRU + grouped coefficient decoder"}],"edges":[["erb","join"],["spec","project"],["project","join"],["join","gru"],["gru","mask"],["gru","coef"],["erb","mask"],["spec","coef"]]},"deepfilter-synthesis":{"title":"Deep filtering and synthesis","scope":"The packaged network predicts a five-frame complex filter for the low-frequency region and a blend factor. Higher bins retain the band-gain result.","blocks":[{"id":"gain","label":"Apply ERB-band gains"},{"id":"filter","label":"Complex multi-frame filtering"},{"id":"blend","label":"Blend filtered / gain-only bins"},{"id":"istft","label":"Inverse FFT + overlap-add"}],"edges":[["gain","filter"],["filter","blend"],["gain","blend"],["blend","istft"]]},"gtcrn":{"title":"GTCRN spectral network","scope":"The dual-path bottleneck separates frequency and temporal recurrence. Decoder skip paths recover spectral detail.","blocks":[{"id":"features","label":"Magnitude + real / imaginary channels"},{"id":"erb","label":"ERB compression + subband unfolding"},{"id":"conv","label":"Convolutional encoder"},{"id":"gt","label":"Grouped temporal blocks","expand":"gtcrn-temporal"},{"id":"rnn","label":"Dual-path grouped GRUs","expand":"gtcrn-dualpath"},{"id":"decode","label":"Grouped temporal decoder + skips","expand":"gtcrn-temporal"},{"id":"out","label":"Frequency upsampling + ERB expansion"}],"edges":[["features","erb"],["erb","conv"],["conv","gt"],["gt","rnn"],["rnn","decode"],["gt","decode"],["conv","out"],["decode","out"]]},"gtcrn-dualpath":{"title":"Dual-path grouped recurrence","scope":"Frequency processing is bidirectional within the current frame. Temporal processing is causal and carries recurrent state.","blocks":[{"id":"intra","label":"Frequency-axis bidirectional grouped GRU"},{"id":"intrap","label":"Linear + LayerNorm + residual"},{"id":"inter","label":"Time-axis grouped GRU"},{"id":"interp","label":"Linear + LayerNorm + residual"}],"edges":[["intra","intrap"],["intrap","inter"],["inter","interp"]]},"gtcrn-temporal":{"title":"Grouped temporal convolution block","scope":"Only half the channels enter the convolutional branch. Recurrent attention gates that branch before shuffling it with the other half.","blocks":[{"id":"split","label":"Split channel groups"},{"id":"conv","label":"Subband / pointwise / depthwise convolutions"},{"id":"attn","label":"Mean-square pooling + GRU + sigmoid"},{"id":"gate","label":"Gate convolutional features"},{"id":"shuffle","label":"Join groups + channel shuffle"}],"edges":[["split","conv"],["conv","attn"],["conv","gate"],["attn","gate"],["gate","shuffle"],["split","shuffle"]]},"complex-mask-synthesis":{"title":"Complex mask reconstruction","scope":"The mask can change both magnitude and phase through complex multiplication.","blocks":[{"id":"mask","label":"Predicted complex mask","kind":"input"},{"id":"spectrum","label":"Input complex spectrum","kind":"input"},{"id":"multiply","label":"Complex multiplication"},{"id":"istft","label":"Inverse STFT / overlap-add"}],"edges":[["mask","multiply"],["spectrum","multiply"],["multiply","istft"]]},"zipenhancer-analysis":{"title":"ZipEnhancer features","scope":"Magnitude compression changes dynamic range without discarding phase.","blocks":[{"id":"norm","label":"Waveform energy normalization"},{"id":"stft","label":"STFT","expand":"complex-stft"},{"id":"mag","label":"Magnitude compression"},{"id":"phase","label":"Phase extraction"},{"id":"complex","label":"Compressed real / imaginary features"}],"edges":[["norm","stft"],["stft","mag"],["stft","phase"],["mag","complex"],["phase","complex"]]},"zipenhancer":{"title":"Dual-path Zipformer enhancement","scope":"Frequency and time Zipformer layers operate at full and reduced resolutions. Learned bypass paths combine down/up-sampled features.","blocks":[{"id":"dense","label":"Dense convolutional encoder"},{"id":"full","label":"Frequency then time Zipformer","expand":"zipformer-layer"},{"id":"down","label":"Time / frequency downsampling"},{"id":"layers","label":"Frequency then time Zipformer","expand":"zipformer-layer"},{"id":"up","label":"Upsampling + learned bypass"},{"id":"last","label":"Full-resolution dual-path Zipformer","expand":"zipformer-layer"}],"edges":[["dense","full"],["full","down"],["down","layers"],["layers","up"],["up","last"]],"residuals":[["down","up"]]},"magnitude-phase-synthesis":{"title":"Magnitude and phase reconstruction","scope":"The phase decoder predicts real and imaginary coordinates; atan2 recovers phase. The waveform's analysis normalization is undone after synthesis.","blocks":[{"id":"features","label":"Enhanced features","kind":"input"},{"id":"mag","label":"Magnitude decoder + decompression"},{"id":"phase","label":"Phase-coordinate decoder + atan2"},{"id":"complex","label":"Assemble complex spectrum"},{"id":"istft","label":"Inverse STFT + rescale"}],"edges":[["features","mag"],["features","phase"],["mag","complex"],["phase","complex"],["complex","istft"]]},"flashsr":{"title":"FlashSR waveform upsampler","scope":"HierSpeech++-derived tiny model, not diffusion FlashSR. Two residual branches use different kernel widths and are averaged.","blocks":[{"id":"conv","label":"Input convolution"},{"id":"up","label":"Linear 3x interpolation"},{"id":"small","label":"Kernel-3 residual branch","expand":"flashsr-resblock"},{"id":"large","label":"Kernel-11 residual branch","expand":"flashsr-resblock"},{"id":"mean","label":"Average branches"},{"id":"act","label":"Alias-filtered Snake activation"},{"id":"out","label":"Output convolution + tanh"}],"edges":[["conv","up"],["up","small"],["up","large"],["small","mean"],["large","mean"],["mean","act"],["act","out"]]},"flashsr-resblock":{"title":"Periodic residual convolution","scope":"Repeated dilated convolution pairs. Each activation upsamples, applies a periodic Snake nonlinearity and low-pass filters before downsampling.","blocks":[{"id":"act1","label":"Alias-filtered Snake activation"},{"id":"conv1","label":"Dilated convolution"},{"id":"act2","label":"Alias-filtered Snake activation"},{"id":"conv2","label":"Convolution"},{"id":"sum","label":"Residual add"}],"edges":[["act1","conv1"],["conv1","act2"],["act2","conv2"],["conv2","sum"]],"residuals":[["act1","sum"]]},"nemotron-diar-encoder":{"title":"Nemotron diarization encoder","scope":"Eight mel frames form one projected encoder frame. The speaker cache and FIFO supply previous context before Transformer encoding, not after it.","blocks":[{"id":"stack","label":"Stack neighboring mel frames"},{"id":"proj","label":"Linear feature projection"},{"id":"context","label":"AOSC / FIFO context assembly","expand":"speaker-cache-context"},{"id":"norm","label":"Embedding LayerNorm"},{"id":"layers","label":"RoPE Transformer layers","expand":"rotary-audio-transformer"},{"id":"final","label":"Final LayerNorm + projection"}],"edges":[["stack","proj"],["proj","context"],["context","norm"],["norm","layers"],["layers","final"]]},"speaker-cache-context":{"title":"Arrival-order speaker context","scope":"State belongs to the session. Speaker activity predictions guide retention of representative past frames for later chunks; recent frames occupy a FIFO queue.","blocks":[{"id":"past","label":"Previous speaker state","kind":"input"},{"id":"current","label":"Current chunk embeddings","kind":"input"},{"id":"cache","label":"Retained speaker-representative frames"},{"id":"fifo","label":"Recent-frame FIFO"},{"id":"join","label":"Concatenate cache / FIFO / chunk"}],"edges":[["past","cache"],["past","fifo"],["cache","join"],["fifo","join"],["current","join"]]},"rotary-audio-transformer":{"title":"Rotary Transformer encoder layer","scope":"Nemotron diarization uses pre-norm attention and a GELU feed-forward network. Its streaming mask is supplied with the chunk and cached context.","blocks":[{"id":"n1","label":"LayerNorm"},{"id":"qkv","label":"Q / K / V projection"},{"id":"attn","label":"RoPE on Q/K + masked self-attention"},{"id":"a1","label":"Output projection + residual"},{"id":"n2","label":"LayerNorm"},{"id":"ff","label":"Linear / GELU / Linear"},{"id":"a2","label":"Residual add"}],"edges":[["n1","qkv"],["qkv","attn"],["attn","a1"],["a1","n2"],["n2","ff"],["ff","a2"]],"residuals":[["n1","a1"],["n2","a2"]]},"nemotron-diar-head":{"title":"Nemotron speaker activity head","scope":"Temporal subpixel upsampling restores the input-feature rate. Sigmoid channels are independent, so multiple speakers may be active at once.","blocks":[{"id":"conv","label":"Temporal convolution"},{"id":"shuffle","label":"Subpixel temporal upsampling"},{"id":"relu1","label":"ReLU + hidden projection"},{"id":"relu2","label":"ReLU + speaker projection"},{"id":"sigmoid","label":"Per-speaker sigmoid"}],"edges":[["conv","shuffle"],["shuffle","relu1"],["relu1","relu2"],["relu2","sigmoid"]]},"sortformer-streaming":{"title":"Streaming Sortformer acoustic encoder","scope":"State is prepended to subsampled features before the Conformer stack. The second Transformer stack and speaker head follow this component.","blocks":[{"id":"sub","label":"Depthwise Conv2D subsampling"},{"id":"proj","label":"Feature projection"},{"id":"context","label":"AOSC / FIFO context assembly","expand":"speaker-cache-context"},{"id":"conformer","label":"Conformer layers","expand":"conformer-block"},{"id":"out","label":"Projection to second encoder"}],"edges":[["sub","proj"],["proj","context"],["context","conformer"],["conformer","out"]]},"sortformer-transformer":{"title":"Sortformer Transformer layer","scope":"Post-norm architecture: normalization follows each residual sum. This is not a causal AR text decoder.","blocks":[{"id":"attn","label":"Masked self-attention"},{"id":"a1","label":"Residual add"},{"id":"n1","label":"LayerNorm"},{"id":"ff","label":"Linear / ReLU / Linear"},{"id":"a2","label":"Residual add"},{"id":"n2","label":"LayerNorm"}],"edges":[["attn","a1"],["a1","n1"],["n1","ff"],["ff","a2"],["a2","n2"]],"residuals":[["attn","a1"],["ff","a2"]]},"sortformer-head":{"title":"Sortformer speaker head","scope":"Speaker channels follow arrival order. Independent probabilities represent overlap without a separate clustering pass.","blocks":[{"id":"relu1","label":"ReLU"},{"id":"hidden","label":"Hidden projection"},{"id":"relu2","label":"ReLU"},{"id":"speaker","label":"Speaker-channel projection"},{"id":"sigmoid","label":"Independent sigmoid probabilities"}],"edges":[["relu1","hidden"],["hidden","relu2"],["relu2","speaker"],["speaker","sigmoid"]]},"speaker-segmentation":{"title":"Speaker turn extraction","scope":"This is interval postprocessing, not a learned speech recognizer. Exact thresholds, smoothing and duration rules belong to the selected model.","blocks":[{"id":"prob","label":"Per-speaker frame probabilities","kind":"input"},{"id":"threshold","label":"Activity thresholding"},{"id":"intervals","label":"Group / refine active intervals"},{"id":"time","label":"Map frames to timestamps"},{"id":"out","label":"Timed speaker turns","kind":"output"}],"edges":[["prob","threshold"],["threshold","intervals"],["intervals","time"],["time","out"]]},"qwen-align-prompt":{"title":"Qwen3 alignment prompt","scope":"The transcript is supplied by the caller, not generated by the aligner. Language controls how words are split.","blocks":[{"id":"normalize","label":"Normalize / split transcript words"},{"id":"placeholders","label":"Two timestamp placeholders per word"},{"id":"tokenize","label":"Tokenize text + audio-slot prompt"}],"edges":[["normalize","placeholders"],["placeholders","tokenize"]]},"qwen-align-head":{"title":"Timestamp-slot classification","scope":"Classification reads the complete supplied prompt in one forward pass. It is not incremental text sampling.","blocks":[{"id":"classify","label":"Linear time-bin classifier"},{"id":"select","label":"Read timestamp placeholder positions"},{"id":"argmax","label":"Choose predicted time bins"},{"id":"repair","label":"Monotonic timestamp repair"},{"id":"out","label":"Word start / end spans","kind":"output"}],"edges":[["classify","select"],["select","argmax"],["argmax","repair"],["repair","out"]]},"wav2vec2-mms":{"title":"MMS Wav2Vec2 encoder","scope":"The MMS aligner uses a pre-norm Wav2Vec2 encoder. It consumes waveform samples, not log-mel features.","blocks":[{"id":"wave","label":"Normalize waveform"},{"id":"conv","label":"Strided Conv1D / LayerNorm / GELU"},{"id":"proj","label":"Feature LayerNorm + projection"},{"id":"pos","label":"Add grouped positional convolution"},{"id":"layers","label":"Bidirectional Transformer layers","expand":"audio-transformer"},{"id":"norm","label":"Final LayerNorm"}],"edges":[["wave","conv"],["conv","proj"],["proj","pos"],["pos","layers"],["layers","norm"]]},"mms-text":{"title":"MMS transcript preparation","scope":"Character mapping is specific to the packaged alignment vocabulary. It creates target labels for dynamic programming, not a language-model prompt.","blocks":[{"id":"text","label":"Known transcript","kind":"input"},{"id":"norm","label":"Normalize / romanize as configured"},{"id":"vocab","label":"Map characters into alignment vocabulary"},{"id":"words","label":"Retain word-to-label boundaries"}],"edges":[["text","norm"],["norm","vocab"],["vocab","words"]]},"ctc-alignment":{"title":"Transcript-constrained CTC alignment","scope":"CTC frame scores are constrained by the supplied target sequence. This differs from greedy CTC transcript decoding.","blocks":[{"id":"scores","label":"Frame log-probabilities","kind":"input"},{"id":"target","label":"Transcript label sequence","kind":"input"},{"id":"dp","label":"Monotonic CTC dynamic program"},{"id":"path","label":"Backtrack label / blank path"},{"id":"merge","label":"Group label spans into words"}],"edges":[["scores","dp"],["target","dp"],["dp","path"],["path","merge"]]},"stft-magnitude":{"title":"Magnitude spectral frontend","scope":"Silero implements Fourier analysis with fixed convolution weights. The resulting real and imaginary channels are combined into magnitudes.","blocks":[{"id":"frame","label":"Context + windowed frames"},{"id":"fourier","label":"Fixed Fourier transform"},{"id":"mag","label":"Magnitude from real / imaginary parts"}],"edges":[["frame","fourier"],["fourier","mag"]]},"silero-encoder":{"title":"Silero CNN + LSTM","scope":"Convolutions compress the spectral frame. The LSTM carries hidden and cell state across successive audio windows.","blocks":[{"id":"conv","label":"Conv1D + ReLU stack"},{"id":"lstm","label":"LSTM cell","expand":"lstm-cell"},{"id":"out","label":"Recurrent hidden features","kind":"output"}],"edges":[["conv","lstm"],["lstm","out"]]},"lstm-cell":{"title":"LSTM cell","scope":"Current features and previous hidden state drive four gates. Cell state preserves a gated memory of previous steps.","blocks":[{"id":"features","label":"Current features","kind":"input"},{"id":"state","label":"Previous hidden / cell state","kind":"input"},{"id":"proj","label":"Input + recurrent projections"},{"id":"gates","label":"Input / forget / output gates + candidate"},{"id":"cell","label":"Gated cell-state update"},{"id":"hidden","label":"Output gate x tanh(cell)"}],"edges":[["features","proj"],["state","proj"],["proj","gates"],["gates","cell"],["state","cell"],["cell","hidden"],["gates","hidden"]]},"silero-head":{"title":"Silero speech probability","scope":"A one-channel head evaluates each recurrent feature frame.","blocks":[{"id":"relu","label":"ReLU"},{"id":"conv","label":"Pointwise speech projection"},{"id":"sigmoid","label":"Sigmoid speech probability"}],"edges":[["relu","conv"],["conv","sigmoid"]]},"marblenet":{"title":"MarbleNet residual block","scope":"Time-channel separable convolution with normalization and ReLU. A projected skip is used in residual blocks; input and output stages may use ordinary convolutions.","blocks":[{"id":"depth","label":"Temporal depthwise convolution"},{"id":"point","label":"Pointwise channel projection"},{"id":"norm","label":"Batch normalization"},{"id":"repeat","label":"ReLU / repeat block stages"},{"id":"add","label":"Projected residual sum + ReLU"}],"edges":[["depth","point"],["point","norm"],["norm","repeat"],["repeat","add"]],"residuals":[["depth","add"]]},"pulsevad":{"title":"PulseVAD tiny CNN","scope":"A compact residual separable CNN over short mel windows; no attention or recurrent state.","blocks":[{"id":"adapter","label":"Pointwise input adapter + ReLU"},{"id":"conv","label":"Depthwise / pointwise convolution"},{"id":"res","label":"Separable residual block"},{"id":"dilated","label":"Dilated depthwise + pointwise stage"},{"id":"pool","label":"Temporal mean pooling"}],"edges":[["adapter","conv"],["conv","res"],["res","dilated"],["dilated","pool"]]},"binary-speech-head":{"title":"Binary speech classifier","scope":"Normalizes speech and non-speech scores. Some MarbleNet checkpoints use a single sigmoid logit; the packaged two-class paths use softmax.","blocks":[{"id":"proj","label":"Speech-class projection"},{"id":"prob","label":"Softmax / sigmoid by checkpoint"},{"id":"out","label":"Speech probability","kind":"output"}],"edges":[["proj","prob"],["prob","out"]]},"vad-segmentation":{"title":"Speech interval extraction","scope":"Model-specific thresholds and duration settings determine boundaries. These are postprocessing controls, not another neural model.","blocks":[{"id":"prob","label":"Speech probabilities","kind":"input"},{"id":"threshold","label":"Threshold / hysteresis policy"},{"id":"duration","label":"Speech / silence duration rules"},{"id":"pad","label":"Optional boundary padding"},{"id":"out","label":"Speech intervals","kind":"output"}],"edges":[["prob","threshold"],["threshold","duration"],["duration","pad"],["pad","out"]]},"vibe-asr-encoders":{"title":"VibeVoice ASR speech encoding","scope":"Two encoders see the same normalized waveform. Acoustic latents and semantic features have separate connectors and are added, not concatenated.","blocks":[{"id":"wave","label":"Normalized waveform","kind":"input"},{"id":"acoustic","label":"Acoustic convolutional encoder","expand":"vibe-tokenizer-block"},{"id":"semantic","label":"Semantic convolutional encoder","expand":"vibe-tokenizer-block"},{"id":"latent","label":"Acoustic latent sampling / scaling"},{"id":"ac","label":"Acoustic connector","expand":"vibe-connector"},{"id":"sc","label":"Semantic connector","expand":"vibe-connector"},{"id":"sum","label":"Elementwise sum at LM width"}],"edges":[["wave","acoustic"],["wave","semantic"],["acoustic","latent"],["latent","ac"],["semantic","sc"],["ac","sum"],["sc","sum"]]},"vibe-tokenizer-block":{"title":"VibeVoice convolutional encoder stage","scope":"Strided causal convolution reduces time resolution. Repeated residual blocks mix time with depthwise convolution and channels with a GELU MLP.","blocks":[{"id":"down","label":"Strided causal convolution"},{"id":"norm1","label":"RMSNorm"},{"id":"mix","label":"Causal depthwise convolution"},{"id":"add1","label":"LayerScale + residual"},{"id":"norm2","label":"RMSNorm"},{"id":"mlp","label":"Linear / GELU / Linear"},{"id":"add2","label":"LayerScale + residual"}],"edges":[["down","norm1"],["norm1","mix"],["mix","add1"],["add1","norm2"],["norm2","mlp"],["mlp","add2"]],"residuals":[["norm1","add1"],["norm2","add2"]]},"vibe-connector":{"title":"Speech-to-LM connector","scope":"Each speech encoder has its own weights for this connector. There is no cross-attention here.","blocks":[{"id":"linear1","label":"Linear projection"},{"id":"norm","label":"RMSNorm"},{"id":"linear2","label":"Linear to LM width"}],"edges":[["linear1","norm"],["norm","linear2"]]},"vibeasr-encoders":{"title":"VibeASR dual encoder","scope":"The packaged quantized path uses ReLU in the encoder MLP. Acoustic and semantic connector outputs are summed before filling speech slots in the LM prompt.","blocks":[{"id":"wave","label":"Normalized waveform","kind":"input"},{"id":"acoustic","label":"Acoustic encoder stages","expand":"vibeasr-encoder-stage"},{"id":"semantic","label":"Semantic encoder stages","expand":"vibeasr-encoder-stage"},{"id":"ac","label":"Acoustic connector","expand":"vibe-connector"},{"id":"sc","label":"Semantic connector","expand":"vibe-connector"},{"id":"sum","label":"Elementwise sum at LM width"}],"edges":[["wave","acoustic"],["wave","semantic"],["acoustic","ac"],["semantic","sc"],["ac","sum"],["sc","sum"]]},"vibeasr-encoder-stage":{"title":"VibeASR quantized encoder stage","scope":"The released I8_S encoder path is distinct from VibeVoice's GELU encoder. Channel mixing uses ReLU and quantized projections.","blocks":[{"id":"down","label":"Strided causal convolution"},{"id":"norm1","label":"RMSNorm"},{"id":"mix","label":"Causal depthwise convolution"},{"id":"add1","label":"LayerScale + residual"},{"id":"norm2","label":"RMSNorm"},{"id":"mlp","label":"Linear / ReLU / Linear"},{"id":"add2","label":"LayerScale + residual"}],"edges":[["down","norm1"],["norm1","mix"],["mix","add1"],["add1","norm2"],["norm2","mlp"],["mlp","add2"]],"residuals":[["norm1","add1"],["norm2","add2"]]},"voxtral-encoder":{"title":"Voxtral causal audio encoder","scope":"Causal convolution and sliding-window rotary attention preserve streaming context. Grouped frames pass through a two-layer GELU adapter.","blocks":[{"id":"conv","label":"Causal Conv1D + GELU frontend"},{"id":"layers","label":"Causal audio Transformer layers","expand":"voxtral-audio-layer"},{"id":"norm","label":"RMSNorm"},{"id":"group","label":"Group consecutive frames"},{"id":"adapter","label":"Linear / GELU / Linear adapter"}],"edges":[["conv","layers"],["layers","norm"],["norm","group"],["group","adapter"]]},"voxtral-audio-layer":{"title":"Voxtral audio Transformer layer","scope":"The audio encoder uses a causal sliding window and rotary positions, not Whisper's bidirectional absolute-position attention.","blocks":[{"id":"n1","label":"RMSNorm"},{"id":"attn","label":"RoPE + causal windowed attention"},{"id":"a1","label":"Residual add"},{"id":"n2","label":"RMSNorm"},{"id":"ff","label":"SwiGLU","expand":"swiglu"},{"id":"a2","label":"Residual add"}],"edges":[["n1","attn"],["attn","a1"],["a1","n2"],["n2","ff"],["ff","a2"]],"residuals":[["n1","a1"],["n2","a2"]]},"voxtral-decoder":{"title":"Voxtral delay-conditioned decoder","scope":"Audio and text embeddings are added before the stack. Every layer modulates its normalized feed-forward input using a learned function of the delay embedding.","blocks":[{"id":"fuse","label":"Audio + token embeddings"},{"id":"n1","label":"RMSNorm"},{"id":"attn","label":"Causal rotary grouped-query attention"},{"id":"a1","label":"Residual add"},{"id":"n2","label":"RMSNorm"},{"id":"delay","label":"Delay embedding / MLP","kind":"input"},{"id":"mod","label":"Multiply by 1 + delay modulation"},{"id":"ff","label":"SwiGLU","expand":"swiglu"},{"id":"a2","label":"Residual add"}],"edges":[["fuse","n1"],["n1","attn"],["attn","a1"],["a1","n2"],["n2","mod"],["delay","mod"],["mod","ff"],["ff","a2"]],"residuals":[["n1","a1"],["n2","a2"]]},"moonshine-frontend":{"title":"Moonshine time-domain frontend","scope":"Waveform frames, not log-mel. Two causal stride-2 convolutions reduce the feature rate.","blocks":[{"id":"frames","label":"Frame waveform"},{"id":"norm","label":"Per-frame mean / RMS normalization"},{"id":"compress","label":"Asinh compression"},{"id":"linear","label":"Linear projection + SiLU"},{"id":"conv","label":"Two causal stride-2 convolutions"}],"edges":[["frames","norm"],["norm","compress"],["compress","linear"],["linear","conv"]]},"moonshine-encoder":{"title":"Moonshine encoder layer","scope":"Windowed attention without positional embeddings. Window lookahead varies across the encoder stack.","blocks":[{"id":"norm1","label":"LayerNorm"},{"id":"attn","label":"Sliding-window self-attention"},{"id":"add1","label":"Residual add"},{"id":"norm2","label":"LayerNorm"},{"id":"mlp","label":"Linear / GELU / Linear"},{"id":"add2","label":"Residual add"}],"edges":[["norm1","attn"],["attn","add1"],["add1","norm2"],["norm2","mlp"],["mlp","add2"]],"residuals":[["norm1","add1"],["norm2","add2"]]},"moonshine-adapter":{"title":"Moonshine positional adapter","scope":"Position information enters after the encoder. A projection is used when encoder and decoder widths differ.","blocks":[{"id":"norm","label":"LayerNorm"},{"id":"pos","label":"Add learned absolute positions"},{"id":"project","label":"Width alignment when needed"}],"edges":[["norm","pos"],["pos","project"]]},"moonshine-decoder":{"title":"Moonshine AR decoder layer","scope":"Text self-attention is causal and rotary; cross-attention reads the encoded audio memory.","blocks":[{"id":"norm1","label":"LayerNorm"},{"id":"self","label":"Causal self-attention + RoPE"},{"id":"add1","label":"Residual add"},{"id":"norm2","label":"LayerNorm"},{"id":"cross","label":"Audio cross-attention"},{"id":"add2","label":"Residual add"},{"id":"norm3","label":"LayerNorm"},{"id":"mlp","label":"SwiGLU feed-forward","expand":"swiglu"},{"id":"add3","label":"Residual add"}],"edges":[["norm1","self"],["self","add1"],["add1","norm2"],["norm2","cross"],["cross","add2"],["add2","norm3"],["norm3","mlp"],["mlp","add3"]],"residuals":[["norm1","add1"],["norm2","add2"],["norm3","add3"]]},"audio8-adapter":{"title":"Audio8 acoustic adapter","scope":"Residual MLP layers precede adaptive average pooling; this is not simply a single linear connector.","blocks":[{"id":"norm","label":"LayerNorm"},{"id":"mlp","label":"Linear / GELU / Linear"},{"id":"add","label":"Residual add; repeat tower layers"},{"id":"final","label":"Final LayerNorm"},{"id":"pool","label":"Adaptive average pooling over time"},{"id":"project","label":"LayerNorm + linear projection"}],"edges":[["norm","mlp"],["mlp","add"],["add","final"],["final","pool"],["pool","project"]],"residuals":[["norm","add"]]},"kaldi-fbank":{"title":"Kaldi-style filterbank frontend","scope":"Checkpoint-specific framing, pre-emphasis and mel frequencies; not an LFR stacking stage.","blocks":[{"id":"frame","label":"Frame / DC removal / pre-emphasis"},{"id":"fft","label":"Window + Fourier transform"},{"id":"power","label":"Power spectrum"},{"id":"mel","label":"Mel filterbank"},{"id":"log","label":"Log energies"}],"edges":[["frame","fft"],["fft","power"],["power","mel"],["mel","log"]]},"zipformer":{"title":"Zipformer acoustic encoder","scope":"Stack-level view: different temporal rates are combined with learned bypass paths. Each stack contains Zipformer layers.","blocks":[{"id":"embed","label":"Convolutional subsampling"},{"id":"down","label":"Temporal downsampling / channel alignment"},{"id":"layers","label":"Zipformer layers","expand":"zipformer-layer"},{"id":"up","label":"Upsampling + learned bypass fusion"},{"id":"out","label":"Encoder output projection"}],"edges":[["embed","down"],["down","layers"],["layers","up"],["up","out"]],"residuals":[["down","up"]]},"zipformer-layer":{"title":"Zipformer layer","scope":"Attention scores are reused by multiple value branches. Convolution and nonlinear attention make this distinct from a plain Transformer.","blocks":[{"id":"ff1","label":"Feed-forward + residual"},{"id":"scores","label":"Relative attention scores"},{"id":"nonlin","label":"Nonlinear attention + residual"},{"id":"attn1","label":"First attention value branch"},{"id":"conv1","label":"First convolution + residual"},{"id":"ff2","label":"Feed-forward + midpoint bypass"},{"id":"attn2","label":"Second attention value branch"},{"id":"conv2","label":"Second convolution + residual"},{"id":"ff3","label":"Feed-forward / BiasNorm / bypass"}],"edges":[["ff1","scores"],["scores","nonlin"],["nonlin","attn1"],["attn1","conv1"],["conv1","ff2"],["ff2","attn2"],["scores","attn2"],["attn2","conv2"],["conv2","ff3"]]},"stateless-transducer":{"title":"Kroko stateless transducer","scope":"Two previous labels condition the predictor; blank emissions advance through acoustic frames.","blocks":[{"id":"tokens","label":"Previous token IDs","kind":"input"},{"id":"embed","label":"Token embeddings"},{"id":"conv","label":"Grouped context convolution + ReLU"},{"id":"proj","label":"Predictor projection"},{"id":"audio","label":"Projected acoustic frame","kind":"input"},{"id":"join","label":"Add + tanh joiner"},{"id":"head","label":"Vocabulary projection + token / blank"}],"edges":[["tokens","embed"],["embed","conv"],["conv","proj"],["proj","join"],["audio","join"],["join","head"]]},"niagara":{"title":"Niagara encoder block","scope":"Niagara is an SSM-plus-attention architecture, not Mamba. Its packaged state-space filter is evaluated as a depthwise temporal convolution.","blocks":[{"id":"ff1","label":"L1-normalized feed-forward + scaled residual"},{"id":"attn","label":"Relative self-attention + residual"},{"id":"ssm","label":"State-space temporal filter","expand":"niagara-ssm"},{"id":"add","label":"Residual add"},{"id":"ff2","label":"Feed-forward + scaled residual"},{"id":"norm","label":"L1 normalization"}],"edges":[["ff1","attn"],["attn","ssm"],["ssm","add"],["add","ff2"],["ff2","norm"]],"residuals":[["ssm","add"]]},"niagara-ssm":{"title":"Niagara state-space sublayer","scope":"The package contains the temporal filter kernel. Channel configuration selects a sigmoid gate or SiLU; this does not use selective Mamba scanning.","blocks":[{"id":"norm","label":"L1 normalization"},{"id":"expand","label":"Channel projection"},{"id":"filter","label":"128-tap depthwise temporal filter"},{"id":"gate","label":"Gating / SiLU"},{"id":"out","label":"Output projection"}],"edges":[["norm","expand"],["expand","filter"],["filter","gate"],["gate","out"]]},"citrinet":{"title":"Citrinet acoustic block","scope":"Time-channel separable convolutions with squeeze-and-excitation and residual paths. No self-attention or language-model decoder is used.","blocks":[{"id":"depth","label":"Depthwise temporal convolution"},{"id":"point","label":"Pointwise channel mixing + BatchNorm"},{"id":"repeat","label":"Repeated conv / activation stages"},{"id":"se","label":"Squeeze-and-excitation"},{"id":"res","label":"Residual sum + activation"}],"edges":[["depth","point"],["point","repeat"],["repeat","se"],["se","res"]],"residuals":[["depth","res"]]},"granite-encoder":{"title":"Granite TurboCTC encoder","scope":"Block-attention Conformer with intermediate CTC self-conditioning. Projection and normalization dimensions remain checkpoint-specific.","blocks":[{"id":"in","label":"Input projection"},{"id":"first","label":"First Conformer layers"},{"id":"ctc","label":"Intermediate CTC distribution"},{"id":"project","label":"Vocabulary-to-hidden projection"},{"id":"add","label":"Add self-conditioning"},{"id":"last","label":"Remaining Conformer layers"}],"edges":[["in","first"],["first","ctc"],["ctc","project"],["project","add"],["add","last"]],"residuals":[["ctc","add"]]},"whisper-encoder":{"title":"Whisper audio encoder","scope":"The audio half of Whisper. Variants differ in width, layer count and output pooling / normalization; no Whisper autoregressive decoder is implied.","blocks":[{"id":"conv","label":"Conv1D / GELU + temporal downsampling"},{"id":"pos","label":"Add absolute positions"},{"id":"stack","label":"Audio Transformer layers","expand":"audio-transformer"},{"id":"out","label":"Audio hidden features","kind":"output"}],"edges":[["conv","pos"],["pos","stack"],["stack","out"]]},"higgs-stt-projector":{"title":"Higgs STT adapter","scope":"The adapter is distinct from Qwen3 and from the Whisper attention stack.","blocks":[{"id":"pool","label":"Average-pool time x2"},{"id":"norm","label":"LayerNorm"},{"id":"conv","label":"Depthwise temporal convolution"},{"id":"mlp","label":"Linear / ReLU / Linear"},{"id":"out","label":"Qwen3 audio embeddings","kind":"output"}],"edges":[["pool","norm"],["norm","conv"],["conv","mlp"],["mlp","out"]]},"moss-stt-projector":{"title":"MOSS transcription adapter","scope":"Continuous encoder features are projected; this path does not sample discrete audio-codec tokens.","blocks":[{"id":"stack","label":"Stack adjacent audio frames"},{"id":"in","label":"Linear projection + SiLU"},{"id":"out","label":"Linear projection"},{"id":"norm","label":"LayerNorm"}],"edges":[["stack","in"],["in","out"],["out","norm"]]},"fbank-lfr":{"title":"Low-frame-rate filterbank frontend","scope":"Windowing, filterbank size, frame stacking and CMVN statistics are checkpoint-specific.","blocks":[{"id":"fbank","label":"Windowed log-filterbank features"},{"id":"lfr","label":"Stack / stride neighboring frames"},{"id":"cmvn","label":"Cepstral mean / variance normalization"}],"edges":[["fbank","lfr"],["lfr","cmvn"]]},"sanm":{"title":"SANM encoder layer","scope":"Self-attention and a sequential-memory filter complement one another; surrounding residual and normalization policies belong to the checkpoint.","blocks":[{"id":"input","label":"Positioned acoustic features","kind":"input"},{"id":"attention","label":"Multi-head self-attention"},{"id":"memory","label":"FSMN temporal memory filter"},{"id":"merge","label":"Combine attention and memory"},{"id":"residual","label":"Residual / normalization"},{"id":"ff","label":"Feed-forward + residual"}],"edges":[["input","attention"],["input","memory"],["attention","merge"],["memory","merge"],["merge","residual"],["residual","ff"]]},"fun-adapter":{"title":"Fun-ASR-Nano audio adapter","scope":"Two adapter layers follow a width-changing MLP. The acoustic encoder and Qwen3 remain separate components.","blocks":[{"id":"project","label":"Linear / ReLU / Linear"},{"id":"attention","label":"Self-attention + residual"},{"id":"ff","label":"ReLU feed-forward + residual"},{"id":"repeat","label":"Repeat adapter layer"},{"id":"out","label":"Qwen3 audio embeddings","kind":"output"}],"edges":[["project","attention"],["attention","ff"],["ff","repeat"],["repeat","out"]]},"samsone-projector":{"title":"SAMSONE pooled projector","scope":"Pooling precedes the residual MLP. The skip connection starts after the first projection, so its width matches the output.","blocks":[{"id":"pool","label":"Temporal average pooling"},{"id":"in","label":"Linear projection"},{"id":"gelu","label":"GELU"},{"id":"out","label":"Linear projection"},{"id":"add","label":"Residual add"},{"id":"norm","label":"LayerNorm"}],"edges":[["pool","in"],["in","gelu"],["gelu","out"],["out","add"],["add","norm"]],"residuals":[["gelu","add"]]},"smollm2-layer":{"title":"SmolLM2 causal decoder layer","scope":"Llama-style RMSNorm, rotary attention and SwiGLU. SAMSONE's checkpoint dimensions and pruned vocabulary are model-specific.","blocks":[{"id":"norm1","label":"RMSNorm"},{"id":"attn","label":"Causal attention + RoPE","expand":"llama-attention"},{"id":"add1","label":"Residual add"},{"id":"norm2","label":"RMSNorm"},{"id":"mlp","label":"SwiGLU","expand":"swiglu"},{"id":"add2","label":"Residual add"}],"edges":[["norm1","attn"],["attn","add1"],["add1","norm2"],["norm2","mlp"],["mlp","add2"]],"residuals":[["norm1","add1"],["norm2","add2"]]},"language-fusion":{"title":"Language-conditioned acoustic features","scope":"Nemotron 3.5's language control enters after FastConformer. It is not an AR text prefix.","blocks":[{"id":"audio","label":"Acoustic frames","kind":"input"},{"id":"language","label":"Language ID","kind":"input"},{"id":"onehot","label":"128-D one-hot / broadcast over time"},{"id":"concat","label":"Concatenate feature channels"},{"id":"project","label":"Projection to decoder input"}],"edges":[["language","onehot"],["audio","concat"],["onehot","concat"],["concat","project"]]},"mel-features":{"title":"Log-mel features","scope":"Signal-processing topology. Sample rate, FFT, mel bins, normalization and streaming context are model-specific.","blocks":[{"id":"prepare","label":"Mono audio / sample-rate handling"},{"id":"stft","label":"Windowed STFT power spectrum"},{"id":"mel","label":"Mel filterbank"},{"id":"log","label":"Log compression + normalization"}],"edges":[["prepare","stft"],["stft","mel"],["mel","log"]]},"conformer-block":{"title":"Conformer block","scope":"Macaron feed-forward branches surround attention and temporal convolution. Position encoding, convolution padding and normalization differ across checkpoints.","blocks":[{"id":"ff1","label":"Pre-norm feed-forward / half residual"},{"id":"attn","label":"Pre-norm self-attention + residual"},{"id":"conv","label":"Temporal convolution + residual","expand":"conformer-conv"},{"id":"ff2","label":"Pre-norm feed-forward / half residual"},{"id":"norm","label":"Final normalization"}],"edges":[["ff1","attn"],["attn","conv"],["conv","ff2"],["ff2","norm"]]},"conformer-conv":{"title":"Conformer convolution module","scope":"Pointwise channel mixing and a depthwise temporal kernel. BatchNorm versus LayerNorm and causal versus symmetric padding are checkpoint-specific.","blocks":[{"id":"norm","label":"Pre-normalization"},{"id":"in","label":"Pointwise projection + GLU"},{"id":"depth","label":"Depthwise temporal convolution"},{"id":"act","label":"Normalization + SiLU"},{"id":"out","label":"Pointwise output projection"},{"id":"add","label":"Residual add"}],"edges":[["norm","in"],["in","depth"],["depth","act"],["act","out"],["out","add"]],"residuals":[["norm","add"]]},"fastconformer":{"title":"Subsampled Conformer encoder","scope":"Convolutional subsampling followed by Conformer blocks. FastConformer, Cohere and Hviske use this high-level topology with separate weights, dimensions and context policies.","blocks":[{"id":"sub","label":"Strided / depthwise Conv2D subsampling"},{"id":"project","label":"Feature projection"},{"id":"stack","label":"Conformer stack","expand":"conformer-block"},{"id":"out","label":"Encoded audio frames","kind":"output"}],"edges":[["sub","project"],["project","stack"],["stack","out"]]},"gigaam-conformer":{"title":"GigaAM rotary Conformer","scope":"Two strided Conv1D layers reduce time by four. Rotary positions are applied before Q/K projection in this model, not after it as in Llama.","blocks":[{"id":"sub","label":"Conv1D subsampling x4"},{"id":"stack","label":"Conformer blocks","expand":"conformer-block"},{"id":"out","label":"Encoded audio frames","kind":"output"}],"edges":[["sub","stack"],["stack","out"]]},"nemo-decoder":{"title":"AR cross-attention decoder","scope":"Canary / Cohere / Hviske decoder topology. Audio supplies cross-attention K/V; text history supplies causal self-attention. This is not a decoder-only LLM.","blocks":[{"id":"self","label":"Norm + causal self-attention"},{"id":"add1","label":"Residual add"},{"id":"audio","label":"Encoded audio memory","kind":"input"},{"id":"cross","label":"Norm + cross-attention"},{"id":"add2","label":"Residual add"},{"id":"ff","label":"Norm + ReLU feed-forward"},{"id":"add3","label":"Residual add"}],"edges":[["self","add1"],["add1","cross"],["audio","cross"],["cross","add2"],["add2","ff"],["ff","add3"]],"residuals":[["self","add1"],["cross","add2"],["ff","add3"]]},"rnnt":{"title":"RNN-T decoding step","scope":"The recurrent predictor advances on emitted nonblank tokens. A blank advances audio time. Edges show one step; decoding repeats with cached recurrent state.","blocks":[{"id":"audio","label":"Encoder frame","kind":"input"},{"id":"token","label":"Previous token + recurrent state","kind":"input"},{"id":"predict","label":"Embedding + LSTM predictor"},{"id":"joint","label":"Joint network"},{"id":"head","label":"Token / blank logits"}],"edges":[["audio","joint"],["token","predict"],["predict","joint"],["joint","head"]]},"tdt":{"title":"Token-and-duration transducer","scope":"The joint network predicts both the token and how many encoder frames to advance. This is distinct from the one-frame blank advance in RNN-T.","blocks":[{"id":"audio","label":"Encoder frame","kind":"input"},{"id":"token","label":"Previous token + recurrent state","kind":"input"},{"id":"predict","label":"Embedding + LSTM predictor"},{"id":"joint","label":"Joint network"},{"id":"head","label":"Token logits"},{"id":"duration","label":"Duration logits"}],"edges":[["audio","joint"],["token","predict"],["predict","joint"],["joint","head"],["joint","duration"]]},"ctc":{"title":"CTC recognition head","scope":"Framewise classification, not autoregressive text generation. Greedy decoding collapses adjacent repeated labels and removes blanks.","blocks":[{"id":"features","label":"Encoder frames","kind":"input"},{"id":"head","label":"Vocabulary + blank projection"},{"id":"decode","label":"Framewise token selection"},{"id":"collapse","label":"Repeat collapse / blank removal"},{"id":"text","label":"Token-to-text decoding"}],"edges":[["features","head"],["head","decode"],["decode","collapse"],["collapse","text"]]},"qwen3-layer":{"title":"Qwen3 decoder layer","scope":"Repeated causal decoder layer. Head counts, width and RoPE configuration belong to the model, not this shared diagram.","blocks":[{"id":"norm1","label":"RMSNorm"},{"id":"attn","label":"Q/K-normalized GQA + RoPE","expand":"qwen3-attention"},{"id":"add1","label":"Residual add"},{"id":"norm2","label":"RMSNorm"},{"id":"mlp","label":"SwiGLU MLP","expand":"swiglu"},{"id":"add2","label":"Residual add"}],"edges":[["norm1","attn"],["attn","add1"],["add1","norm2"],["norm2","mlp"],["mlp","add2"]],"residuals":[["norm1","add1"],["norm2","add2"]]},"qwen2-layer":{"title":"Qwen2 decoder layer","scope":"Qwen2.5 checkpoints use the Qwen2 decoder architecture. Unlike Qwen3, this layer does not add Q/K RMS normalization.","blocks":[{"id":"norm1","label":"RMSNorm"},{"id":"attn","label":"Causal GQA + RoPE","expand":"qwen2-attention"},{"id":"add1","label":"Residual add"},{"id":"norm2","label":"RMSNorm"},{"id":"mlp","label":"SwiGLU MLP","expand":"swiglu"},{"id":"add2","label":"Residual add"}],"edges":[["norm1","attn"],["attn","add1"],["add1","norm2"],["norm2","mlp"],["mlp","add2"]],"residuals":[["norm1","add1"],["norm2","add2"]]},"llama-layer":{"title":"Llama decoder layer","scope":"Pre-normalized causal Transformer. Maya1 uses a Llama architecture; this does not imply a particular Meta checkpoint initialization.","blocks":[{"id":"norm1","label":"RMSNorm"},{"id":"attn","label":"Causal self-attention + RoPE","expand":"llama-attention"},{"id":"add1","label":"Residual add"},{"id":"norm2","label":"RMSNorm"},{"id":"mlp","label":"SwiGLU MLP","expand":"swiglu"},{"id":"add2","label":"Residual add"}],"edges":[["norm1","attn"],["attn","add1"],["add1","norm2"],["norm2","mlp"],["mlp","add2"]],"residuals":[["norm1","add1"],["norm2","add2"]]},"qwen3-attention":{"title":"Qwen3 attention","scope":"Q and K are normalized per head before rotary positions. K/V states are reused during autoregressive decoding.","blocks":[{"id":"project","label":"Q / K / V projections"},{"id":"norm","label":"Q/K RMSNorm"},{"id":"rope","label":"RoPE on Q and K"},{"id":"cache","label":"Append K/V cache"},{"id":"attend","label":"Causal grouped-query attention"},{"id":"output","label":"Output projection"}],"edges":[["project","norm"],["norm","rope"],["rope","cache"],["cache","attend"],["attend","output"]],"residuals":[]},"qwen2-attention":{"title":"Qwen2 attention","scope":"Q/K/V projections include biases. Query heads share fewer key/value heads.","blocks":[{"id":"project","label":"Q / K / V projections + bias"},{"id":"rope","label":"RoPE on Q and K"},{"id":"cache","label":"Append K/V cache"},{"id":"attend","label":"Causal grouped-query attention"},{"id":"output","label":"Output projection"}],"edges":[["project","rope"],["rope","cache"],["cache","attend"],["attend","output"]],"residuals":[]},"llama-attention":{"title":"Llama attention","scope":"Attention head configuration is checkpoint-specific. This schematic omits tensor layout and kernel implementation.","blocks":[{"id":"project","label":"Q / K / V projections"},{"id":"rope","label":"RoPE on Q and K"},{"id":"cache","label":"Append K/V cache"},{"id":"attend","label":"Causal self-attention"},{"id":"output","label":"Output projection"}],"edges":[["project","rope"],["rope","cache"],["cache","attend"],["attend","output"]],"residuals":[]},"swiglu":{"title":"SwiGLU feed-forward network","scope":"The gate and up projections run in parallel; their elementwise product is projected back to the residual width.","blocks":[{"id":"input","label":"Normalized hidden state","kind":"input","rank":0},{"id":"gate","label":"Gate projection + SiLU","rank":1},{"id":"up","label":"Up projection","rank":1},{"id":"mul","label":"Elementwise multiply","rank":2},{"id":"down","label":"Down projection","rank":3}],"edges":[["input","gate"],["input","up"],["gate","mul"],["up","mul"],["mul","down"]],"residuals":[]},"bpe":{"title":"Text tokenization","scope":"Vocabulary, normalization and special tokens are checkpoint-specific. No phoneme conversion is implied.","blocks":[{"id":"format","label":"Prompt / text formatting"},{"id":"split","label":"Pre-tokenization"},{"id":"bpe","label":"Byte-level BPE merges"},{"id":"ids","label":"Token IDs","kind":"output"}],"edges":[["format","split"],["split","bpe"],["bpe","ids"]],"residuals":[]},"soprano-text":{"title":"Soprano text frontend","scope":"The normalizer prepares written text before the model tokenizer. No reference voice branch is used.","blocks":[{"id":"normalize","label":"Text normalization"},{"id":"prompt","label":"Special-token prompt"},{"id":"bpe","label":"BPE tokenizer","expand":"bpe"}],"edges":[["normalize","prompt"],["prompt","bpe"]],"residuals":[]},"logmel":{"title":"Qwen3-ASR audio frontend","scope":"Whisper-style signal processing is not a Whisper neural encoder. Qwen3-ASR uses its own audio Transformer.","blocks":[{"id":"audio","label":"16 kHz mono waveform","kind":"input"},{"id":"stft","label":"Windowed STFT"},{"id":"mel","label":"128-bin mel filterbank"},{"id":"log","label":"Log + dynamic-range scaling"}],"edges":[["audio","stft"],["stft","mel"],["mel","log"]],"residuals":[]},"qwen-audio":{"title":"Qwen3-ASR audio tower","scope":"Convolutional downsampling precedes attention blocks; a projection maps encoded audio into the language-model embedding space.","blocks":[{"id":"conv","label":"Strided Conv2D + GELU"},{"id":"position","label":"Projection + positional embedding"},{"id":"layers","label":"Audio Transformer blocks","expand":"audio-transformer"},{"id":"norm","label":"LayerNorm"},{"id":"project","label":"Output projection"}],"edges":[["conv","position"],["position","layers"],["layers","norm"],["norm","project"]],"residuals":[]},"audio-transformer":{"title":"Audio Transformer encoder layer","scope":"Non-causal audio self-attention follows the encoder window policy, not the AR text decoder mask.","blocks":[{"id":"n1","label":"LayerNorm"},{"id":"attn","label":"Audio self-attention"},{"id":"a1","label":"Residual add"},{"id":"n2","label":"LayerNorm"},{"id":"mlp","label":"Linear / GELU / Linear"},{"id":"a2","label":"Residual add"}],"edges":[["n1","attn"],["attn","a1"],["a1","n2"],["n2","mlp"],["mlp","a2"]],"residuals":[["n1","a1"],["n2","a2"]]},"vocos":{"title":"Vocos waveform synthesis","scope":"The owning model supplies mel frames or generated hidden features. ConvNeXt predicts spectral coefficients for waveform synthesis, not discrete audio tokens.","blocks":[{"id":"project","label":"Feature projection"},{"id":"backbone","label":"ConvNeXt blocks","expand":"convnext"},{"id":"head","label":"Magnitude / phase heads"},{"id":"istft","label":"Inverse STFT","expand":"inverse-stft"},{"id":"wave","label":"Waveform","kind":"output"}],"edges":[["project","backbone"],["backbone","head"],["head","istft"],["istft","wave"]],"residuals":[]},"convnext":{"title":"1D ConvNeXt block","scope":"Conceptual vocoder block. Normalization and activation details may differ between architectures; this diagram is for the Vocos block.","blocks":[{"id":"conv","label":"Depthwise temporal convolution"},{"id":"norm","label":"LayerNorm"},{"id":"up","label":"Pointwise expansion + GELU"},{"id":"down","label":"Pointwise contraction"},{"id":"scale","label":"Layer scale"},{"id":"add","label":"Residual add"}],"edges":[["conv","norm"],["norm","up"],["up","down"],["down","scale"],["scale","add"]],"residuals":[["conv","add"]]},"higgs-reference":{"title":"Higgs reference codec encoder","scope":"Parallel semantic and acoustic features are combined before quantization into the audio prompt codebooks.","blocks":[{"id":"audio","label":"Reference waveform","kind":"input","rank":0},{"id":"semantic","label":"HuBERT semantic encoder","rank":1},{"id":"acoustic","label":"Convolutional audio encoder","rank":1},{"id":"combine","label":"Feature alignment + fusion","rank":2},{"id":"vq","label":"Residual vector quantization","rank":3},{"id":"codes","label":"8-codebook reference prompt","kind":"output","rank":4}],"edges":[["audio","semantic"],["audio","acoustic"],["semantic","combine"],["acoustic","combine"],["combine","vq"],["vq","codes"]],"residuals":[]},"higgs-codec":{"title":"Higgs waveform decoder","scope":"The AR output delay pattern is removed before the codec reconstructs the waveform.","blocks":[{"id":"delay","label":"Undo codebook delay pattern"},{"id":"lookup","label":"Codebook lookup + summation"},{"id":"decoder","label":"Convolutional upsampling decoder"},{"id":"wave","label":"24 kHz waveform","kind":"output"}],"edges":[["delay","lookup"],["lookup","decoder"],["decoder","wave"]],"residuals":[]},"mio-reference":{"title":"MioCodec global reference path","scope":"This conditioning feeds waveform synthesis. It does not feed the MioTTS text-to-content-token language model.","blocks":[{"id":"audio","label":"Reference waveform","kind":"input"},{"id":"wavlm","label":"WavLM features","expand":"wavlm"},{"id":"global","label":"ConvNeXt + attentive pooling","expand":"mio-global"},{"id":"pool","label":"Global reference representation","kind":"output"}],"edges":[["audio","wavlm"],["wavlm","global"],["global","pool"]],"residuals":[]},"miocodec":{"title":"MioCodec waveform synthesis","scope":"FSQ content representations and a global voice condition feed the v2 waveform decoder. No iterative diffusion sampler is involved.","blocks":[{"id":"content","label":"Quantized content representation","kind":"input"},{"id":"ref","label":"Global voice embedding","kind":"input"},{"id":"prenet","label":"Transformer prenet","expand":"mio-transformer"},{"id":"prior","label":"Transpose-conv + residual prior network"},{"id":"decoder","label":"AdaLN-conditioned Transformer","expand":"mio-transformer"},{"id":"up","label":"SnakeBeta residual upsampling"},{"id":"head","label":"Spectral output projection"},{"id":"istft","label":"Inverse STFT","expand":"inverse-stft"}],"edges":[["content","prenet"],["prenet","prior"],["prior","decoder"],["ref","decoder"],["decoder","up"],["up","head"],["head","istft"]],"residuals":[]},"snac":{"title":"SNAC multi-scale decoder","scope":"Maya1 emits interleaved token groups; separate temporal scales are reconstructed before waveform synthesis.","blocks":[{"id":"unpack","label":"Unpack token groups"},{"id":"vq","label":"Multi-rate codebook lookup"},{"id":"align","label":"Upsample + combine latent scales"},{"id":"decoder","label":"Residual convolutional decoder"},{"id":"wave","label":"24 kHz waveform","kind":"output"}],"edges":[["unpack","vq"],["vq","align"],["align","decoder"],["decoder","wave"]],"residuals":[]},"vibe-reference":{"title":"VibeVoice acoustic reference encoder","scope":"The reference encoder produces continuous acoustic latents. It is not a discrete RVQ codec.","blocks":[{"id":"prep","label":"Mono / resample / normalize"},{"id":"conv","label":"Strided causal convolutions"},{"id":"blocks","label":"Depthwise-conv residual blocks"},{"id":"latent","label":"Continuous acoustic latents"},{"id":"project","label":"Acoustic connector to Qwen2"}],"edges":[["prep","conv"],["conv","blocks"],["blocks","latent"],["latent","project"]],"residuals":[]},"vibe-head":{"title":"VibeVoice diffusion head","scope":"A conditioned residual MLP predicts acoustic latents. Calling this head a Transformer DiT would be inaccurate.","blocks":[{"id":"condition","label":"Time + Qwen2 conditioning"},{"id":"project","label":"Noisy latent projection"},{"id":"norm","label":"RMSNorm + adaptive modulation"},{"id":"mlp","label":"Gated SwiGLU residual MLP","expand":"swiglu"},{"id":"out","label":"Modulated output projection"},{"id":"step","label":"Diffusion scheduler step"}],"edges":[["condition","norm"],["project","norm"],["norm","mlp"],["mlp","out"],["out","step"]],"residuals":[]},"vibe-codec":{"title":"VibeVoice acoustic decoder","scope":"Continuous acoustic latents are upsampled through the causal convolutional decoder.","blocks":[{"id":"project","label":"Latent projection"},{"id":"blocks","label":"Residual depthwise-conv blocks"},{"id":"up","label":"Transposed-convolution upsampling"},{"id":"wave","label":"24 kHz waveform","kind":"output"}],"edges":[["project","blocks"],["blocks","up"],["up","wave"]],"residuals":[]}}};