Download Launch_server from neural-nova/GLM-5.2-Instruct-Optimized: direct link, hf CLI and curl.
- Browser
- Download file 1.46 kB
-
https://huggingface.co/neural-nova/GLM-5.2-Instruct-Optimized/resolve/main/Launch_server
- Command line
-
hf download hf://neural-nova/GLM-5.2-Instruct-Optimized/Launch_server
-
curl -L -o Launch_server https://huggingface.co/neural-nova/GLM-5.2-Instruct-Optimized/resolve/main/Launch_server
1.46 kB
| #!/usr/bin/env bash | |
| # Launch vLLM's OpenAI-compatible server for zai-org/GLM-5.2 (ROCm). | |
| # Usage: ./launch_server.sh [baseline|optimized] (default: optimized) | |
| set -euo pipefail | |
| MODE="${1:-optimized}" | |
| MODEL="zai-org/GLM-5.2" | |
| export VLLM_ROCM_USE_AITER=1 | |
| case "$MODE" in | |
| baseline) | |
| # --host/--port added for container networking (omitted in the source | |
| # benchmark notes for this command; see README Notes). | |
| exec vllm serve "$MODEL" \ | |
| --host 0.0.0.0 --port 8000 \ | |
| --tensor-parallel-size 8 \ | |
| --trust-remote-code \ | |
| --gpu-memory-utilization 0.95 \ | |
| --max-model-len 65536 | |
| ;; | |
| optimized) | |
| export VLLM_WORKER_MULTIPROC_METHOD=spawn | |
| # Speculative decoding (MTP) intentionally omitted: it deadlocks on | |
| # gfx942 at TP8 (vllm-project/vllm#48568). | |
| exec vllm serve "$MODEL" \ | |
| --host 0.0.0.0 --port 8000 \ | |
| --tensor-parallel-size 8 \ | |
| --pipeline-parallel-size 1 \ | |
| --distributed-executor-backend mp \ | |
| --max-model-len 65536 \ | |
| --gpu-memory-utilization 0.90 \ | |
| --kv-cache-dtype fp8_e4m3 \ | |
| --max-num-seqs 256 \ | |
| --max-num-batched-tokens 16384 \ | |
| --long-prefill-token-threshold 4096 \ | |
| --enable-prefix-caching \ | |
| --async-scheduling \ | |
| --compilation-config '{"cudagraph_mode": "FULL_AND_PIECEWISE", "max_cudagraph_capture_size": 256}' \ | |
| --trust-remote-code | |
| ;; | |
| *) | |
| echo "Usage: $0 [baseline|optimized]" >&2 | |
| exit 1 | |
| ;; | |
| esac | |