Download gpu-sft/scripts/gpu_sft/deepspeed_zero3_qwen3.json from fzzhang/svd-code: direct link, hf CLI and curl.
- Browser
- Download file 1.77 kB
-
https://huggingface.co/fzzhang/svd-code/resolve/main/gpu-sft/scripts/gpu_sft/deepspeed_zero3_qwen3.json
- Command line
-
hf download hf://fzzhang/svd-code/gpu-sft/scripts/gpu_sft/deepspeed_zero3_qwen3.json
-
curl -L -o deepspeed_zero3_qwen3.json https://huggingface.co/fzzhang/svd-code/resolve/main/gpu-sft/scripts/gpu_sft/deepspeed_zero3_qwen3.json
1.77 kB
| { | |
| "_comment": [ | |
| "Fallback for 40 GB-class GPUs or shops standardised on DeepSpeed. Only the sharding", | |
| "backend changes; the recipe (LR schedule, clipping, loss normalisation, packing) still", | |
| "comes from train_sft_qwen3.py, so run it with --skip-checks and replace the FSDP block", | |
| "with deepspeed.initialize, or drive it through accelerate's DeepSpeed plugin.", | |
| "bf16.enabled with fp32 optimizer states reproduces jmp p=f32,c=bfloat16.", | |
| "gradient_clipping is DISABLED here because the script clips explicitly before step().", | |
| "Optimizer/scheduler are declared 'none' so the script's AdamW + LambdaLR stay in charge." | |
| ], | |
| "train_micro_batch_size_per_gpu": 1, | |
| "gradient_accumulation_steps": 8, | |
| "gradient_clipping": 0.0, | |
| "steps_per_print": 1, | |
| "wall_clock_breakdown": false, | |
| "bf16": { "enabled": true }, | |
| "fp16": { "enabled": false }, | |
| "zero_optimization": { | |
| "stage": 3, | |
| "overlap_comm": true, | |
| "contiguous_gradients": true, | |
| "reduce_bucket_size": 67108864, | |
| "stage3_prefetch_bucket_size": 67108864, | |
| "stage3_param_persistence_threshold": 65536, | |
| "stage3_max_live_parameters": 1000000000, | |
| "stage3_max_reuse_distance": 1000000000, | |
| "stage3_gather_16bit_weights_on_model_save": true, | |
| "offload_optimizer": { "device": "none" }, | |
| "offload_param": { "device": "none" } | |
| }, | |
| "activation_checkpointing": { | |
| "partition_activations": false, | |
| "cpu_checkpointing": false, | |
| "contiguous_memory_optimization": false, | |
| "number_checkpoints": 36, | |
| "synchronize_checkpoint_boundary": false, | |
| "profile": false | |
| }, | |
| "_offload_note": "On 8x40 GB set offload_optimizer.device='cpu' with pin_memory=true: that frees ~12 GB/GPU at a 20-40% throughput cost and does not change the math." | |
| } | |