Download entrypoint.sh from awai-network/basho: direct link, hf CLI and curl.
- Browser
- Download file 1.29 kB
-
https://huggingface.co/awai-network/basho/resolve/main/entrypoint.sh
- Command line
-
hf download hf://awai-network/basho/entrypoint.sh
-
curl -L -o entrypoint.sh https://huggingface.co/awai-network/basho/resolve/main/entrypoint.sh
1.29 kB
| set -euo pipefail | |
| readonly VLLM_ROOT="/usr/local/lib/python3.12/dist-packages/vllm" | |
| readonly MODEL_REPOSITORY="tacos4me/GLM-5.3-Flash-NVFP4-FP8ATTN-512K" | |
| readonly MODEL_REVISION="fd5efd8c534f6170fa9b4143ec57e0eed6faf842" | |
| cp /repository/serving/kda.py "$VLLM_ROOT/models/glm5next/nvidia/kda.py" | |
| cp /repository/serving/model.py "$VLLM_ROOT/models/glm5next/nvidia/model.py" | |
| cp /repository/serving/modelopt.py "$VLLM_ROOT/model_executor/layers/quantization/modelopt.py" | |
| cp /repository/serving/configs/*.json "$VLLM_ROOT/model_executor/layers/quantization/utils/configs/" | |
| exec python3 -m vllm.entrypoints.openai.api_server \ | |
| --model "$MODEL_REPOSITORY" \ | |
| --revision "$MODEL_REVISION" \ | |
| --served-model-name awai-network/basho \ | |
| --host 0.0.0.0 \ | |
| --port 8000 \ | |
| --tensor-parallel-size 2 \ | |
| --max-model-len 16384 \ | |
| --max-num-seqs 1 \ | |
| --max-num-batched-tokens 1024 \ | |
| --kv-cache-dtype fp8 \ | |
| --gpu-memory-utilization 0.95 \ | |
| --kernel-config '{"enable_jit_warmup":false,"enable_cutedsl_warmup":false}' \ | |
| --compilation-config '{"cudagraph_capture_sizes":[1]}' \ | |
| --no-enable-flashinfer-autotune \ | |
| --limit-mm-per-prompt '{"image":0,"video":0}' \ | |
| --trust-remote-code \ | |
| --enable-auto-tool-choice \ | |
| --tool-call-parser glm47 \ | |
| --reasoning-parser glm45 | |