patdev commited on
Commit
a1f8d92
·
verified ·
1 Parent(s): b80e156

Upload k3_bootstrap.sh with huggingface_hub

Browse files
Files changed (1) hide show
  1. k3_bootstrap.sh +12 -2
k3_bootstrap.sh CHANGED
@@ -525,8 +525,18 @@ fi
525
  CPU_EXPERTS='blk\.([1-9]|[12][0-9]|3[0-6]|4[7-9]|[5-7][0-9]|8[0-2])\.ffn_.*_exps'
526
  GPU0_LAYERS='blk\.([0-9]|[1-3][0-9]|4[0-6])\.'
527
  GPU1_LAYERS='blk\.(4[7-9]|[5-8][0-9]|9[0-2])\.'
528
- try "manual balanced placement" "${COMMON[@]}" -ngl 99 --tensor-split 1,1 \
529
- -ot "$CPU_EXPERTS=CPU" -ot "$GPU0_LAYERS=CUDA0" -ot "$GPU1_LAYERS=CUDA1"
 
 
 
 
 
 
 
 
 
 
530
 
531
  # Only 86 is kept: 62, 68, 74 and 80 were each measured OOM-ing on device 1,
532
  # and re-proving that costs a minute of paid GPU apiece on every boot.
 
525
  CPU_EXPERTS='blk\.([1-9]|[12][0-9]|3[0-6]|4[7-9]|[5-7][0-9]|8[0-2])\.ffn_.*_exps'
526
  GPU0_LAYERS='blk\.([0-9]|[1-3][0-9]|4[0-6])\.'
527
  GPU1_LAYERS='blk\.(4[7-9]|[5-8][0-9]|9[0-2])\.'
528
+ # Two-GPU only: the regexes name CUDA0/CUDA1 explicitly and push 72 layers'
529
+ # experts to the host. On a machine with enough VRAM to hold everything, that
530
+ # would deliberately recreate the bottleneck we are trying to remove.
531
+ if [ "$NGPU" = 2 ]; then
532
+ try "manual balanced placement" "${COMMON[@]}" -ngl 99 --tensor-split 1,1 \
533
+ -ot "$CPU_EXPERTS=CPU" -ot "$GPU0_LAYERS=CUDA0" -ot "$GPU1_LAYERS=CUDA1"
534
+ else
535
+ log "skipping manual balanced placement (built for 2 GPUs, this host has $NGPU)"
536
+ # Everything resident, nothing on the host. 194 GB of weights against
537
+ # NGPU x 45 GB; the ggml microbenchmark puts VRAM-resident MoE at 5.5 ms/token.
538
+ try "all-VRAM (-ngl 99, no expert offload)" "${COMMON[@]}" -ngl 99
539
+ fi
540
 
541
  # Only 86 is kept: 62, 68, 74 and 80 were each measured OOM-ing on device 1,
542
  # and re-proving that costs a minute of paid GPU apiece on every boot.