File size: 1,457 Bytes
007e008
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
#!/usr/bin/env bash
# Launch vLLM's OpenAI-compatible server for zai-org/GLM-5.2 (ROCm).
# Usage: ./launch_server.sh [baseline|optimized]   (default: optimized)
set -euo pipefail
MODE="${1:-optimized}"
MODEL="zai-org/GLM-5.2"

export VLLM_ROCM_USE_AITER=1

case "$MODE" in
  baseline)
    # --host/--port added for container networking (omitted in the source
    # benchmark notes for this command; see README Notes).
    exec vllm serve "$MODEL" \
      --host 0.0.0.0 --port 8000 \
      --tensor-parallel-size 8 \
      --trust-remote-code \
      --gpu-memory-utilization 0.95 \
      --max-model-len 65536
    ;;
  optimized)
    export VLLM_WORKER_MULTIPROC_METHOD=spawn
    # Speculative decoding (MTP) intentionally omitted: it deadlocks on
    # gfx942 at TP8 (vllm-project/vllm#48568).
    exec vllm serve "$MODEL" \
      --host 0.0.0.0 --port 8000 \
      --tensor-parallel-size 8 \
      --pipeline-parallel-size 1 \
      --distributed-executor-backend mp \
      --max-model-len 65536 \
      --gpu-memory-utilization 0.90 \
      --kv-cache-dtype fp8_e4m3 \
      --max-num-seqs 256 \
      --max-num-batched-tokens 16384 \
      --long-prefill-token-threshold 4096 \
      --enable-prefix-caching \
      --async-scheduling \
      --compilation-config '{"cudagraph_mode": "FULL_AND_PIECEWISE", "max_cudagraph_capture_size": 256}' \
      --trust-remote-code
    ;;
  *)
    echo "Usage: $0 [baseline|optimized]" >&2
    exit 1
    ;;
esac