File size: 1,767 Bytes
58258b8
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
{
  "_comment": [
    "Fallback for 40 GB-class GPUs or shops standardised on DeepSpeed. Only the sharding",
    "backend changes; the recipe (LR schedule, clipping, loss normalisation, packing) still",
    "comes from train_sft_qwen3.py, so run it with --skip-checks and replace the FSDP block",
    "with deepspeed.initialize, or drive it through accelerate's DeepSpeed plugin.",
    "bf16.enabled with fp32 optimizer states reproduces jmp p=f32,c=bfloat16.",
    "gradient_clipping is DISABLED here because the script clips explicitly before step().",
    "Optimizer/scheduler are declared 'none' so the script's AdamW + LambdaLR stay in charge."
  ],
  "train_micro_batch_size_per_gpu": 1,
  "gradient_accumulation_steps": 8,
  "gradient_clipping": 0.0,
  "steps_per_print": 1,
  "wall_clock_breakdown": false,
  "bf16": { "enabled": true },
  "fp16": { "enabled": false },
  "zero_optimization": {
    "stage": 3,
    "overlap_comm": true,
    "contiguous_gradients": true,
    "reduce_bucket_size": 67108864,
    "stage3_prefetch_bucket_size": 67108864,
    "stage3_param_persistence_threshold": 65536,
    "stage3_max_live_parameters": 1000000000,
    "stage3_max_reuse_distance": 1000000000,
    "stage3_gather_16bit_weights_on_model_save": true,
    "offload_optimizer": { "device": "none" },
    "offload_param": { "device": "none" }
  },
  "activation_checkpointing": {
    "partition_activations": false,
    "cpu_checkpointing": false,
    "contiguous_memory_optimization": false,
    "number_checkpoints": 36,
    "synchronize_checkpoint_boundary": false,
    "profile": false
  },
  "_offload_note": "On 8x40 GB set offload_optimizer.device='cpu' with pin_memory=true: that frees ~12 GB/GPU at a 20-40% throughput cost and does not change the math."
}