SepGen-Generation / config.json
AviadDahan's picture
SepGen Generation checkpoint
7d77286 verified
Raw History Blame Contribute Delete
1.53 kB
{
"model_type": "sepgen",
"adapter_type": "lora",
"base_model": "Lightricks/LTX-2.5",
"base_model_file": "diffusion_models/ltx-2.5-22b-dev-transformer-bf16.safetensors",
"base_model_quantization": "int8-quanto (separation, as in training); bf16 for generation",
"weights_file": "lora_weights.safetensors",
"weights_format": "safetensors, bf16, ComfyUI keys (diffusion_model.*)",
"lora_rank": 128,
"lora_alpha": 128,
"lora_dropout": 0.0,
"lora_modules": 480,
"trainable_parameters": 327155712,
"num_stems": 2,
"audio_span_layout": [
"stem0",
"stem1",
"audio_mix"
],
"code": "https://github.com/AviadDahan/SepGen",
"paper_title": "SepGen: Multi-Stem Audio-Video Separation and Generation in a Single Model",
"license": "LTX-2.x Community License Agreement",
"checkpoint": "gen-3k",
"task": [
"generation",
"separation"
],
"lora_target_modules": [
"audio_attn1.to_q",
"audio_attn1.to_k",
"audio_attn1.to_v",
"audio_attn1.to_out.0",
"audio_attn2.to_q",
"audio_attn2.to_out.0",
"video_to_audio_attn.to_q",
"video_to_audio_attn.to_out.0",
"audio_ff.net.0.proj",
"audio_ff.net.2"
],
"training": {
"steps": 3000,
"init": "sep-12k",
"audio_mix_clean_fraction": 0.3,
"mix_sigma_mode": "independent_leq",
"learning_rate": 0.0001,
"scheduler": "linear to 0.1x over 12000 steps, stopped at step 3000",
"batch_size": 1
},
"sha256": "a72b3c662ce8d6b6e29013bbedb797d48ee1c90e8b7d219af4a27329eb58515b"
}