Download serving/Dockerfile from awai-network/basho: direct link, hf CLI and curl.
- Browser
- Download file 1.25 kB
-
https://huggingface.co/awai-network/basho/resolve/main/serving/Dockerfile
- Command line
-
hf download hf://awai-network/basho/serving/Dockerfile
-
curl -L -o Dockerfile https://huggingface.co/awai-network/basho/resolve/main/serving/Dockerfile
1.25 kB
| # Matched-pair runtime for GLM-5.3-Flash-NVFP4-FP8ATTN-512K. | |
| # Base: the public per-model vLLM image for glm5_next on SM120 (not redistributed here). | |
| # The four COPY lines are the ENTIRE delta — see README.md in this directory. | |
| # | |
| # docker build -t local/vllm-glm53:fp8attn-512k . | |
| FROM cstechdev/vllm:glm53-flash-nope-sm120-cu130-20260826-r1 | |
| # (1) KDA layers: stop stripping quant_config at construction (kda.py:168-175 upstream) | |
| COPY kda.py /usr/local/lib/python3.12/dist-packages/vllm/models/glm5next/nvidia/kda.py | |
| # (2) MLA layers (+ MTP draft): quant_config=None -> real config (model.py:329 upstream) | |
| COPY model.py /usr/local/lib/python3.12/dist-packages/vllm/models/glm5next/nvidia/model.py | |
| # (3) ModelOptMixedPrecisionConfig: FP8_BLOCK128/64/32 dispatch, fused-name resolver, | |
| # MTP draft-prefix aliases, block-FP8 ParallelLMHead loader (vocab-sharded scales) | |
| COPY modelopt.py /usr/local/lib/python3.12/dist-packages/vllm/model_executor/layers/quantization/modelopt.py | |
| # (4) SM120-tuned w8a8 block-FP8 triton config for the KDA fused in_proj shape | |
| # (N=12576, K=4096, block [32,32]) — worth +60% decode on its own | |
| COPY configs/*.json /usr/local/lib/python3.12/dist-packages/vllm/model_executor/layers/quantization/utils/configs/ | |