Image-2.1-Calibrated-NVFP4 / source /verify_native.py
ProCreations's picture
Release calibrated Image2.1 NVFP4 transformer with dynamic scaling and BF16 rank correction, native SM120 runtime, quality evidence and real-time demo
1961af5 verified
Raw History Blame Contribute Delete
1.97 kB
"""Compare packed native FP4 GEMM to explicitly decoded FP32 reference values."""
import json,torch
from pathlib import Path
from safetensors.torch import load_file
from flashinfer.gemm import nvfp4_quantize_smooth,mm_nvfp4_svdquant
ROOT=Path(__file__).parent;OLD=ROOT.with_name('qwen-image-2.1-fp8')
def decode(q,sf):
m,k=q.shape[0],q.shape[1]*2;mp=(m+127)//128*128
s=sf.view(torch.float8_e4m3fn).reshape(mp//128,k//64,32,4,4).permute(0,3,2,1,4).reshape(mp,k//16)[:m].float()
code=torch.stack((q&15,q>>4),-1).reshape(m,k).long()
table=torch.tensor([0.,.5,1.,1.5,2.,3.,4.,6.],device=q.device)
return table[code&7]*torch.where(code&8!=0,-1.,1.)*s.repeat_interleave(16,1)
@torch.inference_mode()
def main():
torch.backends.cuda.matmul.allow_tf32=False;rows=[]
for leaf in ['attn.to_q','img_mlp.proj','img_mlp.out']:
name='transformer_blocks.16.'+leaf
p={k:v.cuda() for k,v in load_file(str(ROOT/'calibrated-layers-dynamic'/f'{name}.safetensors')).items()}
x=load_file(str(OLD/'calibration'/f'{name}.safetensors'))['activations'][:417].cuda().contiguous()
from dynamic_scale import scale
gx,alpha,up=scale(x,p['pre'],p['gx'],p['alpha'],p['up'])
q,sf=nvfp4_quantize_smooth(x,p['pre'],gx,backend='cute-dsl');d=x@p['down']
y=mm_nvfp4_svdquant(q,p['weight'],sf,p['sf'],alpha,d,up,backend='cute-dsl')
ref=(decode(q,sf)@decode(p['weight'],p['sf']).T+d.float()@up.float().T)*alpha
err=((y.float()-ref).square().mean()/ref.square().mean()).sqrt().item()
row={'layer':name,'m':len(x),'relative_nrmse_vs_decoded_fp32':err};rows.append(row);print(row,flush=True);assert err<.004
(ROOT/'native-math-verification.json').write_text(json.dumps({'gpu':torch.cuda.get_device_name(),'torch':torch.__version__,'rows':rows,'pass':True,'method':'Decode actual E2M1 nibbles and E4M3 per16 swizzled scales, multiply in FP32 with TF32 disabled, add exact BF16-down operand times stored correction-up in FP32, compare native fused SM120 FP4 output.'},indent=2))
main()