patdev commited on
Commit
4614aa1
·
verified ·
1 Parent(s): f7dcead

Upload k3_bootstrap.sh with huggingface_hub

Browse files
Files changed (1) hide show
  1. k3_bootstrap.sh +62 -47
k3_bootstrap.sh CHANGED
@@ -356,7 +356,26 @@ log "gpus=$NGPU, letting common_fit_params place everything"
356
  # cudaMalloc OOM, device 0 untouched. With autofit, both cards fill to 95%.
357
  COMMON=(
358
  -m "$FIRST"
359
- --ctx-size 16384
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
360
  --flash-attn on
361
  --jinja
362
  # Not carried by the GGUF metadata; the model mis-routes experts without them.
@@ -374,12 +393,17 @@ COMMON=(
374
  # bottleneck: if it were, halving it would have nearly doubled throughput.
375
  --override-kv "kimi-k3.expert_used_count=int:${K3_EXPERTS:-16}"
376
  --host 0.0.0.0 --port "$PORT"
377
- # Batch-1 decode is the worst case for a MoE model: 16 experts are read from
378
- # VRAM to produce a single token, so the weight traffic is amortised over one
379
- # row. Serving several streams from one forward pass reuses that traffic.
380
- # Slots share --ctx-size (16384/4 = 4096 each), so this costs no extra KV
381
- # VRAM. The CONCURRENCY phase below measures whether it actually pays.
382
- --parallel "${K3_PARALLEL:-4}"
 
 
 
 
 
383
  --threads "$THREADS"
384
  --threads-batch "$THREADS"
385
  # Greedy decoding measurably collapses this ~1-bit quant into repetition
@@ -638,49 +662,40 @@ curl -s -m 900 "http://127.0.0.1:$PORT/v1/chat/completions" -H 'Content-Type: ap
638
  | jq -r 'if .error then "ERROR: \(.error.message)" else (.choices[0].message.content // "EMPTY CONTENT"), (.choices[0].message.reasoning_content // empty) end'
639
  echo
640
 
641
- # ---------------------------------------------------------------- concurrency
642
- # The one question the placement experiments never asked. Six placements moved
643
- # 60-88 GB of weights between CPU and GPU and barely changed single-stream
644
- # throughput; the host vCPU count changed it 1.9x and full VRAM residency
645
- # another 1.4x. What none of them tested is whether the hardware is *idle*
646
- # between tokens.
647
  #
648
- # The discriminator is aggregate throughput vs concurrency:
649
- # scales ~linearly -> decode is latency/serialisation bound (llama.cpp splits
650
- # layers across GPUs, so at batch 1 exactly one GPU
651
- # computes while the others wait). Idle silicon exists
652
- # and concurrency harvests it.
653
- # stays flat -> genuinely bandwidth bound; nothing left on the table
654
- # short of a different quant or fewer GPUs' worth of
655
- # weights.
656
- # Either answer is worth having, and it costs one server start.
657
- echo "===== CONCURRENCY ====="
658
- CONC_PROMPT='Write a detailed explanation of how Mixture-of-Experts routing works:'
659
- CONC_N=64
660
- for C in 1 2 4; do
661
- [ "$C" -gt "${K3_PARALLEL:-4}" ] && continue
662
- t0=$(date +%s.%N)
663
- for i in $(seq 1 "$C"); do
664
- # Distinct prefix per stream: identical prompts would share the prefix
665
- # cache and make prefill look free, inflating the aggregate.
666
- jq -n --arg p "[$i] $CONC_PROMPT" --argjson n "$CONC_N" \
667
- '{prompt:$p,n_predict:$n,temperature:0,repeat_penalty:1.0,stream:false}' \
668
- | curl -s -m 900 "http://127.0.0.1:$PORT/completion" \
669
- -H 'Content-Type: application/json' -d @- >"/tmp/conc.$i.json" &
670
- done
671
- wait
672
- t1=$(date +%s.%N)
673
- gen=0
674
- for i in $(seq 1 "$C"); do
675
- n=$(jq -r '(.timings // {}).predicted_n // 0' "/tmp/conc.$i.json" 2>/dev/null)
676
- gen=$((gen + ${n:-0}))
677
- done
678
- awk -v c="$C" -v g="$gen" -v a="$t0" -v b="$t1" \
679
- 'BEGIN{d=b-a; printf "concurrency %d: %d tok in %.1f s = %.2f tok/s aggregate (%.2f per stream)\n", c, g, d, g/d, g/d/c}'
680
- done
681
  echo
682
 
683
- log "RESULT selftest done - mode=$MODE parallel=${K3_PARALLEL:-4}"
684
  log "server live on port $PORT (/completion and /v1/chat/completions)"
685
 
686
  # Keep serving, and hot-reload on script changes WITHOUT a pod restart.
 
356
  # cudaMalloc OOM, device 0 untouched. With autofit, both cards fill to 95%.
357
  COMMON=(
358
  -m "$FIRST"
359
+ # Long context is unusually cheap here, and the config says why:
360
+ # kv_lora_rank 512 + qk_rope_head_dim 64 = 576 values per token per layer
361
+ # full_attn_layers = every 4th layer -> only 24 of 93 layers cache at all
362
+ # the other 69 are KDA: recurrent state, constant regardless of context
363
+ # So the MLA cache costs 24 * 576 * 2 = 27.6 KB per token:
364
+ # 65536 ctx -> 1.8 GB 262144 -> 7.2 GB 1048576 -> 29 GB
365
+ # against ~38 GB of VRAM left free by the weights on a 5-GPU host. 16384 was
366
+ # never a hardware limit, just an untested default -- and it could not even
367
+ # accept one Claude Code request, whose system prompt alone exceeds 32k.
368
+ # What actually constrains context is the weights sharing the same cards, so
369
+ # check the VRAM dump printed after READY: if the five stop being ~38-40 GB
370
+ # full, autofit has started spilling layers to the CPU and decode collapses.
371
+ --ctx-size "${K3_CTX:-262144}"
372
+ # Prefill is compute-bound and batches, unlike decode. The default micro
373
+ # batch of 512 leaves the GPUs under-fed on the long prompts this client
374
+ # sends; 2048 gives the MoE GEMMs four times the rows to amortise each
375
+ # expert weight read over. It costs compute-buffer VRAM, which is the other
376
+ # thing to watch in that same dump.
377
+ --batch-size "${K3_BATCH:-4096}"
378
+ --ubatch-size "${K3_UBATCH:-2048}"
379
  --flash-attn on
380
  --jinja
381
  # Not carried by the GGUF metadata; the model mis-routes experts without them.
 
393
  # bottleneck: if it were, halving it would have nearly doubled throughput.
394
  --override-kv "kimi-k3.expert_used_count=int:${K3_EXPERTS:-16}"
395
  --host 0.0.0.0 --port "$PORT"
396
+ # Measured on this pod, 64 tokens per stream, same server:
397
+ # 1 stream 5.56 tok/s aggregate
398
+ # 2 streams 9.85 tok/s aggregate (1.77x)
399
+ # 4 streams 14.53 tok/s aggregate (2.61x)
400
+ # Concurrency scales, so decode is bound by serialisation, not bandwidth --
401
+ # at batch 1 the effective read rate is ~51 GB/s against 696 GB/s available
402
+ # on the active card. There is idle silicon and extra streams harvest it.
403
+ # But slots SPLIT --ctx-size, and a Claude Code client needs the whole
404
+ # context for one stream, so the default is 1. Raise K3_PARALLEL (and
405
+ # K3_CTX with it) only when serving several clients at once.
406
+ --parallel "${K3_PARALLEL:-1}"
407
  --threads "$THREADS"
408
  --threads-batch "$THREADS"
409
  # Greedy decoding measurably collapses this ~1-bit quant into repetition
 
662
  | jq -r 'if .error then "ERROR: \(.error.message)" else (.choices[0].message.content // "EMPTY CONTENT"), (.choices[0].message.reasoning_content // empty) end'
663
  echo
664
 
665
+ # ------------------------------------------------------------------- prefill
666
+ # The number that decides whether this serves a Claude Code client. Its system
667
+ # prompt alone is 32k+ tokens, and every selftest prompt above is 6-22 tokens,
668
+ # so their "prefill tok/s" is pure per-request overhead and says nothing.
 
 
669
  #
670
+ # Two costs matter and they are three orders of magnitude apart:
671
+ # cold - the first turn, which must actually run the whole prompt
672
+ # warm - every turn after it, which llama.cpp serves from the slot's prompt
673
+ # cache and re-runs only the tokens that changed
674
+ # Measured on 5x A40, 3408-token prompt: 167 tok/s cold, and the warm repeat
675
+ # reprocessed 4 tokens in 0.53 s. Prefill batches (unlike decode), so it is
676
+ # ~23x the decode rate; a 32k system prompt costs ~3 min once, then ~nothing.
677
+ echo "===== PREFILL ====="
678
+ PF_JSON=/tmp/pf.json
679
+ python3 - "$PF_JSON" "${K3_PF_TOKENS:-32000}" <<'PYEOF'
680
+ import json, sys
681
+ out, target = sys.argv[1], int(sys.argv[2])
682
+ unit = ("Section %d. The routing network assigns each token to a sparse subset "
683
+ "of experts and the gate weights are normalised over the selected set, "
684
+ "which keeps activation cost constant. ")
685
+ # ~34 tokens per unit as tokenised by this model (measured: 100 units = 3408).
686
+ s = "".join(unit % i for i in range(1, max(1, target // 34) + 1))
687
+ s += "
688
+ Summarise the above in one sentence:"
689
+ json.dump({"prompt": s, "n_predict": 4, "temperature": 0, "stream": False}, open(out, "w"))
690
+ PYEOF
691
+ pf_run() {
692
+ curl -s -m 3600 "http://127.0.0.1:$PORT/completion" -H 'Content-Type: application/json' --data-binary @"$PF_JSON" | jq -r '(.timings // {}) | " \(.prompt_n // 0) tok in \(((.prompt_ms // 0)/1000)*10|round/10) s = \((.prompt_per_second // 0)*10|round/10) tok/s"'
693
+ }
694
+ echo "cold (prompt never seen):"; pf_run
695
+ echo "warm (identical prompt, should hit the slot prompt cache):"; pf_run
 
 
 
 
 
 
 
696
  echo
697
 
698
+ log "RESULT selftest done - mode=$MODE parallel=${K3_PARALLEL:-1} ctx=${K3_CTX:-262144}"
699
  log "server live on port $PORT (/completion and /v1/chat/completions)"
700
 
701
  # Keep serving, and hot-reload on script changes WITHOUT a pod restart.