JunYoungLee's picture
Add LoopQ 4-bit quantization of Ouro-1.4B
9118991 verified
Raw History Blame Contribute Delete
1.18 kB
"""Opt-in vLLM V1 worker for packed LoopQ residence and transient BF16 GEMM.
Unlike the diagnostic RPC, installation precedes vLLM's memory profiling so
its transient dequantization allocations participate in the cache budget.
"""
import os
from vllm.v1.worker.gpu_worker import Worker
from loopq.parity_worker import ParityWorkerExtension
class PackedOuroWorker(Worker):
def load_model(self):
directory = os.environ.get('OURO_LOOPQ_PACKED_BUNDLE')
if not directory:
raise ValueError('packed worker requires OURO_LOOPQ_PACKED_BUNDLE')
if self.model_config.enable_sleep_mode:
raise ValueError('packed worker does not support sleep-mode memory pools')
super().load_model()
report = ParityWorkerExtension.loopq_install_packed(
self, directory, allow_diagnostic=False)
# Original loading measured dense QDQ weights. Correct persistent
# allocated bytes before vLLM profiles transient GEMM/activation peaks.
self.model_runner.model_memory_usage += (
report['cuda_allocated_after'] - report['cuda_allocated_before'])
self.loopq_packed_installation = report