"""Regression: nested INT8 offload must move auxiliary tensors and preserve output.""" import argparse import scripts import torch import bitsandbytes as bnb def main(): ap = argparse.ArgumentParser() ap.add_argument('--fixed', action='store_true') args = ap.parse_args() torch.manual_seed(17) layer = bnb.nn.Linear8bitLt(512, 256, has_fp16_weights=False, threshold=6.0).to('cuda') model = torch.nn.Sequential(layer).eval() if args.fixed: from scripts.runtime import patch_int8_device_moves patch_int8_device_moves(model) x = torch.randn(2, 512, device='cuda', dtype=torch.float16) with torch.inference_mode(): expected = model(x).clone() for _ in range(3): model.to('cpu') tensors = [layer.weight, layer.weight.CB, layer.weight.SCB, layer.state.CB, layer.state.SCB, layer.state.idx] assert all(t is None or t.device.type == 'cpu' for t in tensors), 'CUDA tensors retained after parent.to(cpu)' model.to('cuda') with torch.inference_mode(): actual = model(x) torch.testing.assert_close(actual, expected, rtol=0, atol=0) # Also cover the pre-forward quantized parameter CB/SCB path. fresh = torch.nn.Sequential(bnb.nn.Linear8bitLt(512, 256, has_fp16_weights=False).to('cuda')) if args.fixed: patch_int8_device_moves(fresh) fresh.to('cpu') assert fresh[0].weight.CB.device.type == 'cpu' assert fresh[0].weight.SCB.device.type == 'cpu' print('PASS: auxiliary tensors offloaded; three exact-output CPU/GPU round trips') if __name__ == '__main__': main()