Text-to-Speech
English
German
voice-acting
qwen3
moss-audio-tokenizer-v2
audio-generation
Humaneness-Voice-Small / code /large_talker.py
ChristophSchuhmann's picture
Document architecture, prompts, code, and full run statistics
d911efa verified
Raw History Blame Contribute Delete
2.47 kB
#!/usr/bin/env python3
"""Qwen3-0.6B semantic path plus a fresh SFT-3-width local Talker.
This is the existing M1 bridge architecture without loading the SFT-3 Talker
weights and without the experimental K4 memory module. Qwen starts from the
original pretrained Qwen3-0.6B weights; every local/audio/bridge parameter is
fresh and trainable.
"""
from __future__ import annotations
from torch import nn
EXPECTED_LOCAL_TALKER_PARAMETERS = 112_764_928
def build_fresh(schema: dict, log=print):
import moss_small
from conditioning import scored_class
config = moss_small.make_config("M1")
model = scored_class()(config, schema)
moss_small.load_qwen_backbone(model, log=log)
# The score adapter is not part of this caption-conditioned run. It stays
# frozen and receives no score-token prompts, exactly as in the old ladder.
for parameter in model.score_conditioner.parameters():
parameter.requires_grad_(False)
names = (
"local_transformer.", "audio_embeddings.", "local_text_lm_head.",
"proj_in.", "proj_out.",
)
local = sum(parameter.numel() for name, parameter in model.named_parameters()
if name.startswith(names))
assert local == EXPECTED_LOCAL_TALKER_PARAMETERS, local
assert int(config.hidden_size) == 1024
assert int(config.local_hidden_size) == 2560
assert int(config.n_vq) == 12
for index in range(int(config.n_vq)):
assert model.audio_lm_heads[index].weight.data_ptr() == \
model.audio_embeddings[index].weight.data_ptr()
return model, config
def parameter_counts(model: nn.Module) -> dict:
backbone_prefixes = ("transformer.", "text_lm_head.")
backbone = sum(parameter.numel() for name, parameter in model.named_parameters()
if name.startswith(backbone_prefixes) and parameter.requires_grad)
head = sum(parameter.numel() for name, parameter in model.named_parameters()
if not name.startswith(backbone_prefixes) and parameter.requires_grad)
frozen = sum(parameter.numel() for parameter in model.parameters()
if not parameter.requires_grad)
assert head == EXPECTED_LOCAL_TALKER_PARAMETERS, head
return {
"trainable_backbone_parameters": backbone,
"trainable_talker_parameters": head,
"frozen_score_conditioner_parameters": frozen,
"total_parameters": sum(parameter.numel() for parameter in model.parameters()),
}