# HyperPyYAML for the WavLM/FocalCodec hybrid codec. # Checkpoint variant: scratch at 50 Hz. codec_sample_rate: 16000 codec_frequency_hz: 50 # The trained encoder and decoder are provided by the upstream FocalCodec model. focalcodec_source: lucadellalib/focalcodec focalcodec_model: focalcodec focalcodec_config: lucadellalib/focalcodec_50hz focalcodec: !apply:torch.hub.load repo_or_dir: !ref model: !ref config: !ref trust_repo: True pretrained: True encoder: !new:hybridcodec.model.FocalCodecEncoder model: !ref quantizer: !new:hybridcodec.model.SingleLayerResidualQuantizer input_dim: 1024 bottleneck_dim: 32 codebook_size: 8192 semantic_hidden_sizes: [1024, 512, 256] semantic_downscale_factors: [1, 1, 1] residual_hidden_sizes: [512, 256] residual_downscale_factors: [1, 1] continuous_dropout: 0.2 use_post_norm: False vocoder: !ref codec: !new:hybridcodec.model.HybridCodec encoder: !ref quantizer: !ref vocoder: !ref modules: codec: !ref # Pretrainer uses this key as the relative filename under the source repo. pretrainer: !new:speechbrain.utils.parameter_transfer.Pretrainer loadables: "weights/quantizer_scratch_50hz": !ref