Download setup_pod_vaelith.sh from JBARU/lora-training-scripts: direct link, hf CLI and curl.
- Browser
- Download file 3.18 kB
-
https://huggingface.co/JBARU/lora-training-scripts/resolve/main/setup_pod_vaelith.sh
- Command line
-
hf download hf://JBARU/lora-training-scripts/setup_pod_vaelith.sh
-
curl -L -o setup_pod_vaelith.sh https://huggingface.co/JBARU/lora-training-scripts/resolve/main/setup_pod_vaelith.sh
3.18 kB
| # Run this on the RunPod pod terminal (after you've created the pod yourself). | |
| # Assumes an RTX 4090 (24GB) or similar, Ubuntu + CUDA image, network volume at /workspace. | |
| # Network volume should be at least 100GB this time -- kyrael's pod hit disk-full twice at 50GB. | |
| set -e | |
| cd /workspace | |
| echo "=== 1. Clone musubi-tuner ===" | |
| git clone https://github.com/kohya-ss/musubi-tuner | |
| cd musubi-tuner | |
| pip install -e . | |
| pip install transformers accelerate qwen-vl-utils "huggingface_hub[cli]" | |
| echo "=== 2. Log into HuggingFace ===" | |
| echo "Paste your token when prompted -- do NOT put it directly on the command line." | |
| hf auth login | |
| echo "=== 3. Download the RAW Krea2 model (~24.5GB, gated -- must have accepted access on huggingface.co/krea/Krea-2-Raw) ===" | |
| mkdir -p /workspace/models | |
| hf download krea/Krea-2-Raw raw.safetensors --local-dir /workspace/models/krea2_raw | |
| echo "=== 4. Download VAE + text encoder ===" | |
| hf download Comfy-Org/Qwen-Image_ComfyUI split_files/vae/qwen_image_vae.safetensors --local-dir /workspace/models/vae | |
| hf download Comfy-Org/Qwen3-VL text_encoders/qwen3vl_4b_bf16.safetensors --local-dir /workspace/models/text_encoder | |
| echo "=== 5. Download vaelith dataset ===" | |
| mkdir -p /workspace/dataset | |
| hf download JBARU/vaelith-dataset --repo-type dataset --local-dir /workspace/dataset/vaelith | |
| echo "=== 6. Caption dataset (Qwen2.5-VL-7B) ===" | |
| python /workspace/caption_dataset.py /workspace/dataset/vaelith vaelith | |
| echo "=== 6b. Clean up captioning model cache (~16GB) -- this is what caused the disk-full crashes on kyrael's run ===" | |
| rm -rf /workspace/.cache | |
| df -h /workspace | |
| VAE=/workspace/models/vae/split_files/vae/qwen_image_vae.safetensors | |
| TE=/workspace/models/text_encoder/text_encoders/qwen3vl_4b_bf16.safetensors | |
| DIT=/workspace/models/krea2_raw/raw.safetensors | |
| echo "=== 7. Pre-cache latents + text encoder outputs (vaelith) ===" | |
| python src/musubi_tuner/krea2_cache_latents.py --dataset_config /workspace/dataset_vaelith.toml --vae "$VAE" | |
| python src/musubi_tuner/krea2_cache_text_encoder_outputs.py --dataset_config /workspace/dataset_vaelith.toml --text_encoder "$TE" --batch_size 1 | |
| echo "=== 8. Train vaelith LoRA ===" | |
| echo "Using num_repeats=3 this time (kyrael used 10, which caused overtraining/rigidity)." | |
| PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True \ | |
| accelerate launch --num_cpu_threads_per_process 1 --mixed_precision bf16 \ | |
| src/musubi_tuner/krea2_train_network.py \ | |
| --dit "$DIT" --vae "$VAE" \ | |
| --dataset_config /workspace/dataset_vaelith.toml \ | |
| --sdpa --mixed_precision bf16 --fp8_base --fp8_scaled \ | |
| --timestep_sampling shift --weighting_scheme none --discrete_flow_shift 2.5 \ | |
| --optimizer_type adamw8bit --learning_rate 1e-4 --gradient_checkpointing \ | |
| --max_data_loader_n_workers 2 --persistent_data_loader_workers \ | |
| --network_module networks.lora_krea2 --network_dim 32 --network_alpha 16 \ | |
| --max_train_epochs 16 --save_every_n_epochs 2 --seed 42 \ | |
| --output_dir /workspace/output/vaelith --output_name vaelith_lora | |
| echo "=== Done. LoRA is in /workspace/output/vaelith ===" | |
| echo "Back it up immediately with: hf upload <your-username>/vaelith-lora /workspace/output/vaelith --repo-type model --private" | |