{ "model_type": "sepgen", "adapter_type": "lora", "base_model": "Lightricks/LTX-2.5", "base_model_file": "diffusion_models/ltx-2.5-22b-dev-transformer-bf16.safetensors", "base_model_quantization": "int8-quanto (separation, as in training); bf16 for generation", "weights_file": "lora_weights.safetensors", "weights_format": "safetensors, bf16, ComfyUI keys (diffusion_model.*)", "lora_rank": 128, "lora_alpha": 128, "lora_dropout": 0.0, "lora_modules": 480, "trainable_parameters": 327155712, "num_stems": 2, "audio_span_layout": [ "stem0", "stem1", "audio_mix" ], "code": "https://github.com/AviadDahan/SepGen", "paper_title": "SepGen: Multi-Stem Audio-Video Separation and Generation in a Single Model", "license": "LTX-2.x Community License Agreement", "checkpoint": "gen-3k", "task": [ "generation", "separation" ], "lora_target_modules": [ "audio_attn1.to_q", "audio_attn1.to_k", "audio_attn1.to_v", "audio_attn1.to_out.0", "audio_attn2.to_q", "audio_attn2.to_out.0", "video_to_audio_attn.to_q", "video_to_audio_attn.to_out.0", "audio_ff.net.0.proj", "audio_ff.net.2" ], "training": { "steps": 3000, "init": "sep-12k", "audio_mix_clean_fraction": 0.3, "mix_sigma_mode": "independent_leq", "learning_rate": 0.0001, "scheduler": "linear to 0.1x over 12000 steps, stopped at step 3000", "batch_size": 1 }, "sha256": "a72b3c662ce8d6b6e29013bbedb797d48ee1c90e8b7d219af4a27329eb58515b" }