basho / entrypoint.sh
com-junkawasaki's picture
Run Basho on two RTX PRO 6000 GPUs with TP2
bd9eeec verified
Raw History Blame Contribute Delete
1.29 kB
#!/usr/bin/env bash
set -euo pipefail
readonly VLLM_ROOT="/usr/local/lib/python3.12/dist-packages/vllm"
readonly MODEL_REPOSITORY="tacos4me/GLM-5.3-Flash-NVFP4-FP8ATTN-512K"
readonly MODEL_REVISION="fd5efd8c534f6170fa9b4143ec57e0eed6faf842"
cp /repository/serving/kda.py "$VLLM_ROOT/models/glm5next/nvidia/kda.py"
cp /repository/serving/model.py "$VLLM_ROOT/models/glm5next/nvidia/model.py"
cp /repository/serving/modelopt.py "$VLLM_ROOT/model_executor/layers/quantization/modelopt.py"
cp /repository/serving/configs/*.json "$VLLM_ROOT/model_executor/layers/quantization/utils/configs/"
exec python3 -m vllm.entrypoints.openai.api_server \
--model "$MODEL_REPOSITORY" \
--revision "$MODEL_REVISION" \
--served-model-name awai-network/basho \
--host 0.0.0.0 \
--port 8000 \
--tensor-parallel-size 2 \
--max-model-len 16384 \
--max-num-seqs 1 \
--max-num-batched-tokens 1024 \
--kv-cache-dtype fp8 \
--gpu-memory-utilization 0.95 \
--kernel-config '{"enable_jit_warmup":false,"enable_cutedsl_warmup":false}' \
--compilation-config '{"cudagraph_capture_sizes":[1]}' \
--no-enable-flashinfer-autotune \
--limit-mm-per-prompt '{"image":0,"video":0}' \
--trust-remote-code \
--enable-auto-tool-choice \
--tool-call-parser glm47 \
--reasoning-parser glm45