NeuralNovaAI's picture
Create Launch_server
007e008 verified
Raw History Blame Contribute Delete
1.46 kB
#!/usr/bin/env bash
# Launch vLLM's OpenAI-compatible server for zai-org/GLM-5.2 (ROCm).
# Usage: ./launch_server.sh [baseline|optimized] (default: optimized)
set -euo pipefail
MODE="${1:-optimized}"
MODEL="zai-org/GLM-5.2"
export VLLM_ROCM_USE_AITER=1
case "$MODE" in
baseline)
# --host/--port added for container networking (omitted in the source
# benchmark notes for this command; see README Notes).
exec vllm serve "$MODEL" \
--host 0.0.0.0 --port 8000 \
--tensor-parallel-size 8 \
--trust-remote-code \
--gpu-memory-utilization 0.95 \
--max-model-len 65536
;;
optimized)
export VLLM_WORKER_MULTIPROC_METHOD=spawn
# Speculative decoding (MTP) intentionally omitted: it deadlocks on
# gfx942 at TP8 (vllm-project/vllm#48568).
exec vllm serve "$MODEL" \
--host 0.0.0.0 --port 8000 \
--tensor-parallel-size 8 \
--pipeline-parallel-size 1 \
--distributed-executor-backend mp \
--max-model-len 65536 \
--gpu-memory-utilization 0.90 \
--kv-cache-dtype fp8_e4m3 \
--max-num-seqs 256 \
--max-num-batched-tokens 16384 \
--long-prefill-token-threshold 4096 \
--enable-prefix-caching \
--async-scheduling \
--compilation-config '{"cudagraph_mode": "FULL_AND_PIECEWISE", "max_cudagraph_capture_size": 256}' \
--trust-remote-code
;;
*)
echo "Usage: $0 [baseline|optimized]" >&2
exit 1
;;
esac