File size: 1,637 Bytes
ba48d54
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
"""Regression: nested INT8 offload must move auxiliary tensors and preserve output."""
import argparse
import scripts
import torch
import bitsandbytes as bnb


def main():
    ap = argparse.ArgumentParser()
    ap.add_argument('--fixed', action='store_true')
    args = ap.parse_args()
    torch.manual_seed(17)
    layer = bnb.nn.Linear8bitLt(512, 256, has_fp16_weights=False, threshold=6.0).to('cuda')
    model = torch.nn.Sequential(layer).eval()
    if args.fixed:
        from scripts.runtime import patch_int8_device_moves
        patch_int8_device_moves(model)
    x = torch.randn(2, 512, device='cuda', dtype=torch.float16)
    with torch.inference_mode():
        expected = model(x).clone()
    for _ in range(3):
        model.to('cpu')
        tensors = [layer.weight, layer.weight.CB, layer.weight.SCB,
                   layer.state.CB, layer.state.SCB, layer.state.idx]
        assert all(t is None or t.device.type == 'cpu' for t in tensors), 'CUDA tensors retained after parent.to(cpu)'
        model.to('cuda')
        with torch.inference_mode():
            actual = model(x)
        torch.testing.assert_close(actual, expected, rtol=0, atol=0)
    # Also cover the pre-forward quantized parameter CB/SCB path.
    fresh = torch.nn.Sequential(bnb.nn.Linear8bitLt(512, 256, has_fp16_weights=False).to('cuda'))
    if args.fixed:
        patch_int8_device_moves(fresh)
    fresh.to('cpu')
    assert fresh[0].weight.CB.device.type == 'cpu'
    assert fresh[0].weight.SCB.device.type == 'cpu'
    print('PASS: auxiliary tensors offloaded; three exact-output CPU/GPU round trips')


if __name__ == '__main__':
    main()