File size: 1,767 Bytes
58258b8 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 | {
"_comment": [
"Fallback for 40 GB-class GPUs or shops standardised on DeepSpeed. Only the sharding",
"backend changes; the recipe (LR schedule, clipping, loss normalisation, packing) still",
"comes from train_sft_qwen3.py, so run it with --skip-checks and replace the FSDP block",
"with deepspeed.initialize, or drive it through accelerate's DeepSpeed plugin.",
"bf16.enabled with fp32 optimizer states reproduces jmp p=f32,c=bfloat16.",
"gradient_clipping is DISABLED here because the script clips explicitly before step().",
"Optimizer/scheduler are declared 'none' so the script's AdamW + LambdaLR stay in charge."
],
"train_micro_batch_size_per_gpu": 1,
"gradient_accumulation_steps": 8,
"gradient_clipping": 0.0,
"steps_per_print": 1,
"wall_clock_breakdown": false,
"bf16": { "enabled": true },
"fp16": { "enabled": false },
"zero_optimization": {
"stage": 3,
"overlap_comm": true,
"contiguous_gradients": true,
"reduce_bucket_size": 67108864,
"stage3_prefetch_bucket_size": 67108864,
"stage3_param_persistence_threshold": 65536,
"stage3_max_live_parameters": 1000000000,
"stage3_max_reuse_distance": 1000000000,
"stage3_gather_16bit_weights_on_model_save": true,
"offload_optimizer": { "device": "none" },
"offload_param": { "device": "none" }
},
"activation_checkpointing": {
"partition_activations": false,
"cpu_checkpointing": false,
"contiguous_memory_optimization": false,
"number_checkpoints": 36,
"synchronize_checkpoint_boundary": false,
"profile": false
},
"_offload_note": "On 8x40 GB set offload_optimizer.device='cpu' with pin_memory=true: that frees ~12 GB/GPU at a 20-40% throughput cost and does not change the math."
}
|