File size: 1,251 Bytes
9e9e40a
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
# Matched-pair runtime for GLM-5.3-Flash-NVFP4-FP8ATTN-512K.
# Base: the public per-model vLLM image for glm5_next on SM120 (not redistributed here).
# The four COPY lines are the ENTIRE delta — see README.md in this directory.
#
#   docker build -t local/vllm-glm53:fp8attn-512k .
FROM cstechdev/vllm:glm53-flash-nope-sm120-cu130-20260826-r1

# (1) KDA layers: stop stripping quant_config at construction (kda.py:168-175 upstream)
COPY kda.py   /usr/local/lib/python3.12/dist-packages/vllm/models/glm5next/nvidia/kda.py
# (2) MLA layers (+ MTP draft): quant_config=None -> real config (model.py:329 upstream)
COPY model.py /usr/local/lib/python3.12/dist-packages/vllm/models/glm5next/nvidia/model.py
# (3) ModelOptMixedPrecisionConfig: FP8_BLOCK128/64/32 dispatch, fused-name resolver,
#     MTP draft-prefix aliases, block-FP8 ParallelLMHead loader (vocab-sharded scales)
COPY modelopt.py /usr/local/lib/python3.12/dist-packages/vllm/model_executor/layers/quantization/modelopt.py
# (4) SM120-tuned w8a8 block-FP8 triton config for the KDA fused in_proj shape
#     (N=12576, K=4096, block [32,32]) — worth +60% decode on its own
COPY configs/*.json /usr/local/lib/python3.12/dist-packages/vllm/model_executor/layers/quantization/utils/configs/