{ "_comment": [ "Fallback for 40 GB-class GPUs or shops standardised on DeepSpeed. Only the sharding", "backend changes; the recipe (LR schedule, clipping, loss normalisation, packing) still", "comes from train_sft_qwen3.py, so run it with --skip-checks and replace the FSDP block", "with deepspeed.initialize, or drive it through accelerate's DeepSpeed plugin.", "bf16.enabled with fp32 optimizer states reproduces jmp p=f32,c=bfloat16.", "gradient_clipping is DISABLED here because the script clips explicitly before step().", "Optimizer/scheduler are declared 'none' so the script's AdamW + LambdaLR stay in charge." ], "train_micro_batch_size_per_gpu": 1, "gradient_accumulation_steps": 8, "gradient_clipping": 0.0, "steps_per_print": 1, "wall_clock_breakdown": false, "bf16": { "enabled": true }, "fp16": { "enabled": false }, "zero_optimization": { "stage": 3, "overlap_comm": true, "contiguous_gradients": true, "reduce_bucket_size": 67108864, "stage3_prefetch_bucket_size": 67108864, "stage3_param_persistence_threshold": 65536, "stage3_max_live_parameters": 1000000000, "stage3_max_reuse_distance": 1000000000, "stage3_gather_16bit_weights_on_model_save": true, "offload_optimizer": { "device": "none" }, "offload_param": { "device": "none" } }, "activation_checkpointing": { "partition_activations": false, "cpu_checkpointing": false, "contiguous_memory_optimization": false, "number_checkpoints": 36, "synchronize_checkpoint_boundary": false, "profile": false }, "_offload_note": "On 8x40 GB set offload_optimizer.device='cpu' with pin_memory=true: that frees ~12 GB/GPU at a 20-40% throughput cost and does not change the math." }