ProCreations's picture
Accelerate full 40-step FP8 generation with native precision, measured quality and real-time demo
1081be0 verified
Raw
History Blame Contribute Delete
1.02 kB
{
"load_seconds": 2.426407459017355,
"torch": "2.14.0+cu130",
"gpu": "NVIDIA RTX PRO 6000 Blackwell Workstation Edition",
"fp8_linears": 224,
"steps": 40,
"cfg": 1,
"extra_quantization": false,
"approximate_cache": false,
"timing": {
"1024": {
"seconds": [
5.903008527995553,
5.9249284460092895,
5.935668507008813,
5.946647913951892,
5.951301124005113
],
"mean": 5.932310903794132,
"warmup_seconds": 11.555044600041583,
"peak_gb": 32.590829568
},
"2048": {
"seconds": [
37.25538438500371,
37.301657135016285,
37.31008757499512
],
"mean": 37.2890430316717,
"warmup_seconds": 39.20627289195545,
"peak_gb": 53.722432512
}
},
"protocol": "CUDA synchronized; batch1; full40steps; includes encoder, denoising and VAE; excludes model load, resolution warmup and file writes. Weights and prefixKVcache unchanged. Compiled mode emulates intermediate precision casts."
}