Image-2.1-Calibrated-FP8 / optimization /baseline-benchmark.json
ProCreations's picture
Accelerate full 40-step FP8 generation with native precision, measured quality and real-time demo
1081be0 verified
Raw History Blame Contribute Delete
915 Bytes
{
"load_seconds": 2.426567278977018,
"torch": "2.14.0+cu130",
"gpu": "NVIDIA RTX PRO 6000 Blackwell Workstation Edition",
"fp8_linears": 224,
"steps": 40,
"cfg": 1,
"extra_quantization": false,
"approximate_cache": false,
"timing": {
"1024": {
"seconds": [
6.9087469020159915,
6.936133738025092
],
"mean": 6.9224403200205415,
"warmup_seconds": 8.378067673009355,
"peak_gb": 32.590829568
},
"2048": {
"seconds": [
44.079785163048655,
44.1227304089698
],
"mean": 44.10125778600923,
"warmup_seconds": 43.912163228029385,
"peak_gb": 53.722432512
}
},
"protocol": "CUDA synchronized; batch1; full40steps; includes encoder, denoising and VAE; excludes model load, resolution warmup and file writes. Weights and prefixKVcache unchanged. Compiled mode emulates intermediate precision casts."
}