Image21-INT8 / tests /gpu_offload.py
ixim's picture
Fix INT8 CPU offload; publish memory audit and paired retests
ba48d54 verified
Raw History Blame Contribute Delete
1.64 kB
"""Regression: nested INT8 offload must move auxiliary tensors and preserve output."""
import argparse
import scripts
import torch
import bitsandbytes as bnb
def main():
ap = argparse.ArgumentParser()
ap.add_argument('--fixed', action='store_true')
args = ap.parse_args()
torch.manual_seed(17)
layer = bnb.nn.Linear8bitLt(512, 256, has_fp16_weights=False, threshold=6.0).to('cuda')
model = torch.nn.Sequential(layer).eval()
if args.fixed:
from scripts.runtime import patch_int8_device_moves
patch_int8_device_moves(model)
x = torch.randn(2, 512, device='cuda', dtype=torch.float16)
with torch.inference_mode():
expected = model(x).clone()
for _ in range(3):
model.to('cpu')
tensors = [layer.weight, layer.weight.CB, layer.weight.SCB,
layer.state.CB, layer.state.SCB, layer.state.idx]
assert all(t is None or t.device.type == 'cpu' for t in tensors), 'CUDA tensors retained after parent.to(cpu)'
model.to('cuda')
with torch.inference_mode():
actual = model(x)
torch.testing.assert_close(actual, expected, rtol=0, atol=0)
# Also cover the pre-forward quantized parameter CB/SCB path.
fresh = torch.nn.Sequential(bnb.nn.Linear8bitLt(512, 256, has_fp16_weights=False).to('cuda'))
if args.fixed:
patch_int8_device_moves(fresh)
fresh.to('cpu')
assert fresh[0].weight.CB.device.type == 'cpu'
assert fresh[0].weight.SCB.device.type == 'cpu'
print('PASS: auxiliary tensors offloaded; three exact-output CPU/GPU round trips')
if __name__ == '__main__':
main()