File size: 1,251 Bytes
9e9e40a | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 | # Matched-pair runtime for GLM-5.3-Flash-NVFP4-FP8ATTN-512K.
# Base: the public per-model vLLM image for glm5_next on SM120 (not redistributed here).
# The four COPY lines are the ENTIRE delta — see README.md in this directory.
#
# docker build -t local/vllm-glm53:fp8attn-512k .
FROM cstechdev/vllm:glm53-flash-nope-sm120-cu130-20260826-r1
# (1) KDA layers: stop stripping quant_config at construction (kda.py:168-175 upstream)
COPY kda.py /usr/local/lib/python3.12/dist-packages/vllm/models/glm5next/nvidia/kda.py
# (2) MLA layers (+ MTP draft): quant_config=None -> real config (model.py:329 upstream)
COPY model.py /usr/local/lib/python3.12/dist-packages/vllm/models/glm5next/nvidia/model.py
# (3) ModelOptMixedPrecisionConfig: FP8_BLOCK128/64/32 dispatch, fused-name resolver,
# MTP draft-prefix aliases, block-FP8 ParallelLMHead loader (vocab-sharded scales)
COPY modelopt.py /usr/local/lib/python3.12/dist-packages/vllm/model_executor/layers/quantization/modelopt.py
# (4) SM120-tuned w8a8 block-FP8 triton config for the KDA fused in_proj shape
# (N=12576, K=4096, block [32,32]) — worth +60% decode on its own
COPY configs/*.json /usr/local/lib/python3.12/dist-packages/vllm/model_executor/layers/quantization/utils/configs/
|