SepGen-Separation / config.json
AviadDahan's picture
SepGen Separation checkpoint
a09b692 verified
Raw History Blame Contribute Delete
1.48 kB
{
"model_type": "sepgen",
"adapter_type": "lora",
"base_model": "Lightricks/LTX-2.5",
"base_model_file": "diffusion_models/ltx-2.5-22b-dev-transformer-bf16.safetensors",
"base_model_quantization": "int8-quanto (separation, as in training); bf16 for generation",
"weights_file": "lora_weights.safetensors",
"weights_format": "safetensors, bf16, ComfyUI keys (diffusion_model.*)",
"lora_rank": 128,
"lora_alpha": 128,
"lora_dropout": 0.0,
"lora_modules": 480,
"trainable_parameters": 327155712,
"num_stems": 2,
"audio_span_layout": [
"stem0",
"stem1",
"audio_mix"
],
"code": "https://github.com/AviadDahan/SepGen",
"paper_title": "SepGen: Multi-Stem Audio-Video Separation and Generation in a Single Model",
"license": "LTX-2.x Community License Agreement",
"checkpoint": "sep-12k",
"task": [
"separation"
],
"lora_target_modules": [
"audio_attn1.to_q",
"audio_attn1.to_k",
"audio_attn1.to_v",
"audio_attn1.to_out.0",
"audio_attn2.to_q",
"audio_attn2.to_out.0",
"video_to_audio_attn.to_q",
"video_to_audio_attn.to_out.0",
"audio_ff.net.0.proj",
"audio_ff.net.2"
],
"training": {
"steps": 12000,
"init": null,
"audio_mix_clean_fraction": 1.0,
"mix_sigma_mode": "independent",
"learning_rate": 0.0001,
"scheduler": "linear to 0.1x over 12000 steps",
"batch_size": 1
},
"sha256": "fc0987e972c0e92556348ea99b018c9a7c3a95691c4bd46a7d549b1fcd4edd39"
}