File size: 1,290 Bytes
7b4ba28
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
bd9eeec
7b4ba28
 
 
 
 
11826d2
bd9eeec
11826d2
7b4ba28
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
#!/usr/bin/env bash
set -euo pipefail

readonly VLLM_ROOT="/usr/local/lib/python3.12/dist-packages/vllm"
readonly MODEL_REPOSITORY="tacos4me/GLM-5.3-Flash-NVFP4-FP8ATTN-512K"
readonly MODEL_REVISION="fd5efd8c534f6170fa9b4143ec57e0eed6faf842"

cp /repository/serving/kda.py "$VLLM_ROOT/models/glm5next/nvidia/kda.py"
cp /repository/serving/model.py "$VLLM_ROOT/models/glm5next/nvidia/model.py"
cp /repository/serving/modelopt.py "$VLLM_ROOT/model_executor/layers/quantization/modelopt.py"
cp /repository/serving/configs/*.json "$VLLM_ROOT/model_executor/layers/quantization/utils/configs/"

exec python3 -m vllm.entrypoints.openai.api_server \
  --model "$MODEL_REPOSITORY" \
  --revision "$MODEL_REVISION" \
  --served-model-name awai-network/basho \
  --host 0.0.0.0 \
  --port 8000 \
  --tensor-parallel-size 2 \
  --max-model-len 16384 \
  --max-num-seqs 1 \
  --max-num-batched-tokens 1024 \
  --kv-cache-dtype fp8 \
  --gpu-memory-utilization 0.95 \
  --kernel-config '{"enable_jit_warmup":false,"enable_cutedsl_warmup":false}' \
  --compilation-config '{"cudagraph_capture_sizes":[1]}' \
  --no-enable-flashinfer-autotune \
  --limit-mm-per-prompt '{"image":0,"video":0}' \
  --trust-remote-code \
  --enable-auto-tool-choice \
  --tool-call-parser glm47 \
  --reasoning-parser glm45