basho / serving /Dockerfile
com-junkawasaki's picture
Add pinned GLM-5.3 vLLM runtime patches
9e9e40a verified
Raw History Blame Contribute Delete
1.25 kB
# Matched-pair runtime for GLM-5.3-Flash-NVFP4-FP8ATTN-512K.
# Base: the public per-model vLLM image for glm5_next on SM120 (not redistributed here).
# The four COPY lines are the ENTIRE delta — see README.md in this directory.
#
# docker build -t local/vllm-glm53:fp8attn-512k .
FROM cstechdev/vllm:glm53-flash-nope-sm120-cu130-20260826-r1
# (1) KDA layers: stop stripping quant_config at construction (kda.py:168-175 upstream)
COPY kda.py /usr/local/lib/python3.12/dist-packages/vllm/models/glm5next/nvidia/kda.py
# (2) MLA layers (+ MTP draft): quant_config=None -> real config (model.py:329 upstream)
COPY model.py /usr/local/lib/python3.12/dist-packages/vllm/models/glm5next/nvidia/model.py
# (3) ModelOptMixedPrecisionConfig: FP8_BLOCK128/64/32 dispatch, fused-name resolver,
# MTP draft-prefix aliases, block-FP8 ParallelLMHead loader (vocab-sharded scales)
COPY modelopt.py /usr/local/lib/python3.12/dist-packages/vllm/model_executor/layers/quantization/modelopt.py
# (4) SM120-tuned w8a8 block-FP8 triton config for the KDA fused in_proj shape
# (N=12576, K=4096, block [32,32]) — worth +60% decode on its own
COPY configs/*.json /usr/local/lib/python3.12/dist-packages/vllm/model_executor/layers/quantization/utils/configs/