Download loopq_quantization/scripts/loopq/packed_worker.py from JunYoungLee/ut-depth-probe-artifacts: direct link, hf CLI and curl.
- Browser
- Download file 1.18 kB
-
https://huggingface.co/JunYoungLee/ut-depth-probe-artifacts/resolve/main/loopq_quantization/scripts/loopq/packed_worker.py
- Command line
-
hf download hf://JunYoungLee/ut-depth-probe-artifacts/loopq_quantization/scripts/loopq/packed_worker.py
-
curl -L -o packed_worker.py https://huggingface.co/JunYoungLee/ut-depth-probe-artifacts/resolve/main/loopq_quantization/scripts/loopq/packed_worker.py
1.18 kB
| """Opt-in vLLM V1 worker for packed LoopQ residence and transient BF16 GEMM. | |
| Unlike the diagnostic RPC, installation precedes vLLM's memory profiling so | |
| its transient dequantization allocations participate in the cache budget. | |
| """ | |
| import os | |
| from vllm.v1.worker.gpu_worker import Worker | |
| from loopq.parity_worker import ParityWorkerExtension | |
| class PackedOuroWorker(Worker): | |
| def load_model(self): | |
| directory = os.environ.get('OURO_LOOPQ_PACKED_BUNDLE') | |
| if not directory: | |
| raise ValueError('packed worker requires OURO_LOOPQ_PACKED_BUNDLE') | |
| if self.model_config.enable_sleep_mode: | |
| raise ValueError('packed worker does not support sleep-mode memory pools') | |
| super().load_model() | |
| report = ParityWorkerExtension.loopq_install_packed( | |
| self, directory, allow_diagnostic=False) | |
| # Original loading measured dense QDQ weights. Correct persistent | |
| # allocated bytes before vLLM profiles transient GEMM/activation peaks. | |
| self.model_runner.model_memory_usage += ( | |
| report['cuda_allocated_after'] - report['cuda_allocated_before']) | |
| self.loopq_packed_installation = report | |