#!/usr/bin/env bash set -euo pipefail readonly VLLM_ROOT="/usr/local/lib/python3.12/dist-packages/vllm" readonly MODEL_REPOSITORY="tacos4me/GLM-5.3-Flash-NVFP4-FP8ATTN-512K" readonly MODEL_REVISION="fd5efd8c534f6170fa9b4143ec57e0eed6faf842" cp /repository/serving/kda.py "$VLLM_ROOT/models/glm5next/nvidia/kda.py" cp /repository/serving/model.py "$VLLM_ROOT/models/glm5next/nvidia/model.py" cp /repository/serving/modelopt.py "$VLLM_ROOT/model_executor/layers/quantization/modelopt.py" cp /repository/serving/configs/*.json "$VLLM_ROOT/model_executor/layers/quantization/utils/configs/" exec python3 -m vllm.entrypoints.openai.api_server \ --model "$MODEL_REPOSITORY" \ --revision "$MODEL_REVISION" \ --served-model-name awai-network/basho \ --host 0.0.0.0 \ --port 8000 \ --tensor-parallel-size 2 \ --max-model-len 16384 \ --max-num-seqs 1 \ --max-num-batched-tokens 1024 \ --kv-cache-dtype fp8 \ --gpu-memory-utilization 0.95 \ --kernel-config '{"enable_jit_warmup":false,"enable_cutedsl_warmup":false}' \ --compilation-config '{"cudagraph_capture_sizes":[1]}' \ --no-enable-flashinfer-autotune \ --limit-mm-per-prompt '{"image":0,"video":0}' \ --trust-remote-code \ --enable-auto-tool-choice \ --tool-call-parser glm47 \ --reasoning-parser glm45