Instructions to use ProCreations/Image-2.1-Calibrated-FP8 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Diffusers
How to use ProCreations/Image-2.1-Calibrated-FP8 with Diffusers:
pip install -U diffusers transformers accelerate
import torch from diffusers import DiffusionPipeline # switch to "mps" for apple devices pipe = DiffusionPipeline.from_pretrained("ProCreations/Image-2.1-Calibrated-FP8", dtype=torch.bfloat16, device_map="cuda") prompt = "Astronaut in a jungle, cold color palette, muted colors, detailed, 8k" image = pipe(prompt).images[0] - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- Draw Things
- DiffusionBee
Accelerate full 40-step FP8 generation with native precision, measured quality and real-time demo
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- .gitattributes +34 -0
- README.md +18 -1
- acceleration.py +91 -0
- demo/optimized-capture_receipt.json +0 -0
- demo/optimized-realtime-30s.mp4 +3 -0
- generate.py +32 -9
- manifest.json +270 -5
- optimization/REPORT.md +35 -0
- optimization/baseline-benchmark.json +31 -0
- optimization/benchmark.json +35 -0
- optimization/comparisons/00.jpg +3 -0
- optimization/comparisons/01.jpg +3 -0
- optimization/comparisons/02.jpg +3 -0
- optimization/comparisons/03.jpg +3 -0
- optimization/comparisons/04.jpg +3 -0
- optimization/comparisons/05.jpg +3 -0
- optimization/comparisons/06.jpg +3 -0
- optimization/comparisons/07.jpg +3 -0
- optimization/comparisons/08.jpg +3 -0
- optimization/comparisons/09.jpg +0 -0
- optimization/comparisons/10.jpg +0 -0
- optimization/comparisons/11.jpg +3 -0
- optimization/comparisons/12.jpg +3 -0
- optimization/comparisons/13.jpg +0 -0
- optimization/comparisons/14.jpg +3 -0
- optimization/comparisons/15.jpg +3 -0
- optimization/comparisons/edit-0.jpg +3 -0
- optimization/comparisons/edit-1.jpg +3 -0
- optimization/denoising-trace.json +0 -0
- optimization/kernel_evidence.json +75 -0
- optimization/qa-review.json +16 -0
- optimization/quality-vs-bf16.json +168 -0
- optimization/quality-vs-fp8.json +168 -0
- optimization/samples/00.png +3 -0
- optimization/samples/01.png +3 -0
- optimization/samples/02.png +3 -0
- optimization/samples/03.png +3 -0
- optimization/samples/04.png +3 -0
- optimization/samples/05.png +3 -0
- optimization/samples/06.png +3 -0
- optimization/samples/07.png +3 -0
- optimization/samples/08.png +3 -0
- optimization/samples/09.png +3 -0
- optimization/samples/10.png +3 -0
- optimization/samples/11.png +3 -0
- optimization/samples/12.png +3 -0
- optimization/samples/13.png +3 -0
- optimization/samples/14.png +3 -0
- optimization/samples/15.png +3 -0
- optimization/samples/edit-0.png +3 -0
.gitattributes
CHANGED
|
@@ -73,3 +73,37 @@ samples/14.png filter=lfs diff=lfs merge=lfs -text
|
|
| 73 |
samples/15.png filter=lfs diff=lfs merge=lfs -text
|
| 74 |
samples/edit-0.png filter=lfs diff=lfs merge=lfs -text
|
| 75 |
samples/edit-1.png filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 73 |
samples/15.png filter=lfs diff=lfs merge=lfs -text
|
| 74 |
samples/edit-0.png filter=lfs diff=lfs merge=lfs -text
|
| 75 |
samples/edit-1.png filter=lfs diff=lfs merge=lfs -text
|
| 76 |
+
demo/optimized-realtime-30s.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 77 |
+
optimization/comparisons/00.jpg filter=lfs diff=lfs merge=lfs -text
|
| 78 |
+
optimization/comparisons/01.jpg filter=lfs diff=lfs merge=lfs -text
|
| 79 |
+
optimization/comparisons/02.jpg filter=lfs diff=lfs merge=lfs -text
|
| 80 |
+
optimization/comparisons/03.jpg filter=lfs diff=lfs merge=lfs -text
|
| 81 |
+
optimization/comparisons/04.jpg filter=lfs diff=lfs merge=lfs -text
|
| 82 |
+
optimization/comparisons/05.jpg filter=lfs diff=lfs merge=lfs -text
|
| 83 |
+
optimization/comparisons/06.jpg filter=lfs diff=lfs merge=lfs -text
|
| 84 |
+
optimization/comparisons/07.jpg filter=lfs diff=lfs merge=lfs -text
|
| 85 |
+
optimization/comparisons/08.jpg filter=lfs diff=lfs merge=lfs -text
|
| 86 |
+
optimization/comparisons/11.jpg filter=lfs diff=lfs merge=lfs -text
|
| 87 |
+
optimization/comparisons/12.jpg filter=lfs diff=lfs merge=lfs -text
|
| 88 |
+
optimization/comparisons/14.jpg filter=lfs diff=lfs merge=lfs -text
|
| 89 |
+
optimization/comparisons/15.jpg filter=lfs diff=lfs merge=lfs -text
|
| 90 |
+
optimization/comparisons/edit-0.jpg filter=lfs diff=lfs merge=lfs -text
|
| 91 |
+
optimization/comparisons/edit-1.jpg filter=lfs diff=lfs merge=lfs -text
|
| 92 |
+
optimization/samples/00.png filter=lfs diff=lfs merge=lfs -text
|
| 93 |
+
optimization/samples/01.png filter=lfs diff=lfs merge=lfs -text
|
| 94 |
+
optimization/samples/02.png filter=lfs diff=lfs merge=lfs -text
|
| 95 |
+
optimization/samples/03.png filter=lfs diff=lfs merge=lfs -text
|
| 96 |
+
optimization/samples/04.png filter=lfs diff=lfs merge=lfs -text
|
| 97 |
+
optimization/samples/05.png filter=lfs diff=lfs merge=lfs -text
|
| 98 |
+
optimization/samples/06.png filter=lfs diff=lfs merge=lfs -text
|
| 99 |
+
optimization/samples/07.png filter=lfs diff=lfs merge=lfs -text
|
| 100 |
+
optimization/samples/08.png filter=lfs diff=lfs merge=lfs -text
|
| 101 |
+
optimization/samples/09.png filter=lfs diff=lfs merge=lfs -text
|
| 102 |
+
optimization/samples/10.png filter=lfs diff=lfs merge=lfs -text
|
| 103 |
+
optimization/samples/11.png filter=lfs diff=lfs merge=lfs -text
|
| 104 |
+
optimization/samples/12.png filter=lfs diff=lfs merge=lfs -text
|
| 105 |
+
optimization/samples/13.png filter=lfs diff=lfs merge=lfs -text
|
| 106 |
+
optimization/samples/14.png filter=lfs diff=lfs merge=lfs -text
|
| 107 |
+
optimization/samples/15.png filter=lfs diff=lfs merge=lfs -text
|
| 108 |
+
optimization/samples/edit-0.png filter=lfs diff=lfs merge=lfs -text
|
| 109 |
+
optimization/samples/edit-1.png filter=lfs diff=lfs merge=lfs -text
|
README.md
CHANGED
|
@@ -36,7 +36,22 @@ Calibration prompts and seeds are published in [the manifest](reports/calibratio
|
|
| 36 |
|
| 37 |
Evaluation uses 16 separate generation prompts and 2 separate edits, matched seeds, 40 steps, the same prefix KV-cache setting and the same BF16 encoder/VAE. At 512 px for perceptual comparison, mean LPIPS(AlexNet) is **0.0345** (lower is closer), mean SSIM is **0.9713** (higher is closer). Mean full-resolution final-latent cosine similarity is **0.995627**. These measure agreement with the BF16 reference; they do not establish a broad human-preference score or guarantee every prompt. See [all paired comparisons](comparisons/) and [per-image metrics](reports/quality_metrics.json). Rendering may change under a different kernel, precision, dependency version or KV-cache setting.
|
| 38 |
|
| 39 |
-
##
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 40 |
|
| 41 |
RTX PRO 6000 Blackwell 96 GB; batch 1; 40 steps; CFG 1; full GPU residency; prefix KV cache enabled. CUDA-synchronized wall time includes prompt encoding, denoising and VAE decode; excludes model loading and PNG writing. One warmup per resolution is excluded; 1024 has 5 measured repeats and 2048 has 3. No reduced-step adapter or timestep skipping is used.
|
| 42 |
|
|
@@ -61,6 +76,8 @@ python generate.py --prompt "A kingfisher above a forest stream, wildlife photog
|
|
| 61 |
|
| 62 |
Download this repository first with `hf download ProCreations/Image-2.1-Calibrated-FP8 --local-dir Image-2.1-Calibrated-FP8`, then run the commands from its directory. The loader fetches the unmodified pipeline components from upstream revision `b3179ad355be050328e483a9dfdd9e60cd62adfa`. To reuse a complete local upstream snapshot, pass `--base /path/to/snapshot`. For editing, pass `--image input.png`. Use explicit dimensions and 40 steps to match the benchmark.
|
| 63 |
|
|
|
|
|
|
|
| 64 |
Source model: `Qwen/Qwen-Image-2.1@b3179ad355be050328e483a9dfdd9e60cd62adfa`.
|
| 65 |
Diffusers: `80c7ed262aeffbeb43ef13ae04baeb9b84515a69`.
|
| 66 |
|
|
|
|
| 36 |
|
| 37 |
Evaluation uses 16 separate generation prompts and 2 separate edits, matched seeds, 40 steps, the same prefix KV-cache setting and the same BF16 encoder/VAE. At 512 px for perceptual comparison, mean LPIPS(AlexNet) is **0.0345** (lower is closer), mean SSIM is **0.9713** (higher is closer). Mean full-resolution final-latent cosine similarity is **0.995627**. These measure agreement with the BF16 reference; they do not establish a broad human-preference score or guarantee every prompt. See [all paired comparisons](comparisons/) and [per-image metrics](reports/quality_metrics.json). Rendering may change under a different kernel, precision, dependency version or KV-cache setting.
|
| 38 |
|
| 39 |
+
## Full-compute runtime update
|
| 40 |
+
|
| 41 |
+
The default `generate.py` now uses **decode-only dynamic compilation and kernel fusion**, keeping all 40 denoising steps, original calibrated FP8 weights/scales, FP32 GEMM accumulation, and native BF16 attention. No step reuse, distilled adapter, further attention quantization, or reduced resolution is enabled. `--eager` selects the original runtime.
|
| 42 |
+
|
| 43 |
+
| Resolution | Original FP8, fresh control | Optimized FP8 | Speedup |
|
| 44 |
+
|---|---:|---:|---:|
|
| 45 |
+
| 1024×1024 | 6.922 s | 5.932 s | 1.167× |
|
| 46 |
+
| 2048×2048 | 44.101 s | 37.289 s | 1.183× |
|
| 47 |
+
|
| 48 |
+
Same RTX PRO 6000, batch 1, 40 steps, CFG 1, prefix KV cache. CUDA-synchronized end-to-end generation includes encoding and VAE. Warmup/model loading/PNG writes excluded. Original control has 2 measured repeats per size; optimized has 5 at 1024 and 3 at 2048. [Raw measurements](optimization/benchmark.json) include first-use warmup and load costs. First-use compilation costs extra time. `--warmup` reports that cost separately; `--prompts-json prompts.json` processes a list of prompt strings in one loaded pipeline.
|
| 49 |
+
|
| 50 |
+
All 18 paired checks were visually reviewed, including primary English/Chinese titles, portraits, transparency and edits. No obvious general visual-quality loss was observed in this finite set. **Outputs are not bit-identical**: fine textures, decorative typography and some pottery positioning change with GPU reduction rounding. Mean LPIPS versus original FP8=0.014105, mean SSIM=0.985841; worst LPIPS=0.104154 (pottery). Relative to BF16, mean LPIPS=0.035437, compared with 0.034470 for the original FP8. These are fidelity checks, not a guarantee for every prompt. [Comparisons](optimization/comparisons/) · [Detailed report](optimization/REPORT.md).
|
| 51 |
+
|
| 52 |
+
[Updated 30-second real-time image-only video](demo/optimized-realtime-30s.mp4). One completed warmup image is shown at time 0, then each new image appears when actual generation finishes. All waits remain at 1× speed. [Frame/request timestamps](demo/optimized-capture_receipt.json).
|
| 53 |
+
|
| 54 |
+
## Initial uncompiled latency
|
| 55 |
|
| 56 |
RTX PRO 6000 Blackwell 96 GB; batch 1; 40 steps; CFG 1; full GPU residency; prefix KV cache enabled. CUDA-synchronized wall time includes prompt encoding, denoising and VAE decode; excludes model loading and PNG writing. One warmup per resolution is excluded; 1024 has 5 measured repeats and 2048 has 3. No reduced-step adapter or timestep skipping is used.
|
| 57 |
|
|
|
|
| 76 |
|
| 77 |
Download this repository first with `hf download ProCreations/Image-2.1-Calibrated-FP8 --local-dir Image-2.1-Calibrated-FP8`, then run the commands from its directory. The loader fetches the unmodified pipeline components from upstream revision `b3179ad355be050328e483a9dfdd9e60cd62adfa`. To reuse a complete local upstream snapshot, pass `--base /path/to/snapshot`. For editing, pass `--image input.png`. Use explicit dimensions and 40 steps to match the benchmark.
|
| 78 |
|
| 79 |
+
Library use: call `accelerate_pipeline(pipe)` from `acceleration.py` after `load_pipeline(...)`. The underlying `load_pipeline` function retains its original eager behavior. The default command uses compilation.
|
| 80 |
+
|
| 81 |
Source model: `Qwen/Qwen-Image-2.1@b3179ad355be050328e483a9dfdd9e60cd62adfa`.
|
| 82 |
Diffusers: `80c7ed262aeffbeb43ef13ae04baeb9b84515a69`.
|
| 83 |
|
acceleration.py
ADDED
|
@@ -0,0 +1,91 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Full-compute acceleration for Image 2.1 Calibrated FP8. Built with Qwen.
|
| 2 |
+
|
| 3 |
+
Keeps every denoising step, BF16 attention, original calibrated FP8 GEMMs and
|
| 4 |
+
FP32 accumulation/scales. No approximate residual cache or attention quantization.
|
| 5 |
+
Only cached-prefix decode blocks compile; prefill keeps upstream behavior.
|
| 6 |
+
"""
|
| 7 |
+
import types
|
| 8 |
+
import torch
|
| 9 |
+
from diffusers.models.transformers.transformer_qwenimage21 import (
|
| 10 |
+
QwenImage21AttnProcessor, QwenImage21TransformerBlock,
|
| 11 |
+
)
|
| 12 |
+
|
| 13 |
+
_ORIGINAL_BLOCK_FORWARD = QwenImage21TransformerBlock.forward
|
| 14 |
+
|
| 15 |
+
|
| 16 |
+
def _real_rope(x, frequencies):
|
| 17 |
+
paired = x.float().unflatten(-1, (-1, 2))
|
| 18 |
+
cosine = frequencies.real[None, :, None, :]
|
| 19 |
+
sine = frequencies.imag[None, :, None, :]
|
| 20 |
+
return torch.stack((
|
| 21 |
+
paired[..., 0] * cosine - paired[..., 1] * sine,
|
| 22 |
+
paired[..., 0] * sine + paired[..., 1] * cosine,
|
| 23 |
+
), dim=-1).flatten(-2).to(x.dtype)
|
| 24 |
+
|
| 25 |
+
|
| 26 |
+
class NativeAttentionProcessor(QwenImage21AttnProcessor):
|
| 27 |
+
def __call__(self, attn, hidden_states, attention_mask=None, rotary_emb=None,
|
| 28 |
+
layer_cache=None, kv_cache_mode=None, cache_write_slice=None,
|
| 29 |
+
segments=None, key_valid=None):
|
| 30 |
+
if kv_cache_mode != 'cached' or attention_mask is not None:
|
| 31 |
+
return super().__call__(attn, hidden_states, attention_mask, rotary_emb,
|
| 32 |
+
layer_cache, kv_cache_mode, cache_write_slice,
|
| 33 |
+
segments, key_valid)
|
| 34 |
+
query = attn.to_q(hidden_states).unflatten(-1, (attn.heads, -1))
|
| 35 |
+
key = attn.to_k(hidden_states).unflatten(-1, (attn.heads, -1))
|
| 36 |
+
value = attn.to_v(hidden_states).unflatten(-1, (attn.heads, -1))
|
| 37 |
+
query = attn.norm_q(query)
|
| 38 |
+
key = attn.norm_k(key)
|
| 39 |
+
if rotary_emb is not None:
|
| 40 |
+
query = _real_rope(query, rotary_emb)
|
| 41 |
+
key = _real_rope(key, rotary_emb)
|
| 42 |
+
cached_key, cached_value = layer_cache.get()
|
| 43 |
+
key = torch.cat((cached_key, key), dim=1)
|
| 44 |
+
value = torch.cat((cached_value, value), dim=1)
|
| 45 |
+
output = torch.nn.functional.scaled_dot_product_attention(
|
| 46 |
+
query.transpose(1, 2), key.transpose(1, 2), value.transpose(1, 2)
|
| 47 |
+
).transpose(1, 2)
|
| 48 |
+
return attn.to_out[1](attn.to_out[0](output.flatten(2, 3)))
|
| 49 |
+
|
| 50 |
+
|
| 51 |
+
def _decode_block(self, hidden_states, modulation, rotary_emb=None,
|
| 52 |
+
attention_mask=None, target_token_mask=None, layer_cache=None,
|
| 53 |
+
kv_cache_mode=None, cache_write_slice=None, segments=None,
|
| 54 |
+
key_valid=None):
|
| 55 |
+
# Cached decode contains only target image tokens. The final t=0 modulation
|
| 56 |
+
# row belongs to the prefix, which was already evaluated during prefill.
|
| 57 |
+
if kv_cache_mode == 'cached':
|
| 58 |
+
modulation = modulation[:-1]
|
| 59 |
+
target_token_mask = None
|
| 60 |
+
return _ORIGINAL_BLOCK_FORWARD(
|
| 61 |
+
self, hidden_states, modulation, rotary_emb, attention_mask,
|
| 62 |
+
target_token_mask, layer_cache, kv_cache_mode, cache_write_slice,
|
| 63 |
+
segments, key_valid,
|
| 64 |
+
)
|
| 65 |
+
|
| 66 |
+
|
| 67 |
+
def _dispatch_block(self, **kwargs):
|
| 68 |
+
if kwargs.get('kv_cache_mode') == 'cached':
|
| 69 |
+
return self._image21_compiled(**kwargs)
|
| 70 |
+
return _ORIGINAL_BLOCK_FORWARD(self, **kwargs)
|
| 71 |
+
|
| 72 |
+
|
| 73 |
+
def accelerate_pipeline(pipe):
|
| 74 |
+
"""Enable once after loading. Initial compilation is excluded from warm timings.
|
| 75 |
+
|
| 76 |
+
Dynamic sequence lengths reduce recompilation across prompts/resolutions.
|
| 77 |
+
emulate_precision_casts preserves the upstream intermediate BF16 rounding
|
| 78 |
+
boundaries in fused code; GPU reduction ordering can still differ.
|
| 79 |
+
"""
|
| 80 |
+
if getattr(pipe, '_image21_accelerated', False):
|
| 81 |
+
return pipe
|
| 82 |
+
torch._dynamo.config.recompile_limit = max(torch._dynamo.config.recompile_limit, 64)
|
| 83 |
+
for block in pipe.transformer.transformer_blocks:
|
| 84 |
+
block.attn.set_processor(NativeAttentionProcessor())
|
| 85 |
+
block._image21_compiled = torch.compile(
|
| 86 |
+
types.MethodType(_decode_block, block), fullgraph=True, dynamic=True,
|
| 87 |
+
options={'emulate_precision_casts': True},
|
| 88 |
+
)
|
| 89 |
+
block.forward = types.MethodType(_dispatch_block, block)
|
| 90 |
+
pipe._image21_accelerated = True
|
| 91 |
+
return pipe
|
demo/optimized-capture_receipt.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
demo/optimized-realtime-30s.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:3271da65a4797dda30419d7857612fb05873497f2c50b18056c96fde92fb9457
|
| 3 |
+
size 1753321
|
generate.py
CHANGED
|
@@ -1,12 +1,15 @@
|
|
| 1 |
"""Generate with Image 2.1 Calibrated FP8. Built with Qwen."""
|
| 2 |
-
import argparse,json,time
|
| 3 |
from pathlib import Path
|
| 4 |
import torch
|
| 5 |
from fp8_runtime import load_pipeline
|
| 6 |
|
|
|
|
| 7 |
def main():
|
| 8 |
ap=argparse.ArgumentParser()
|
| 9 |
-
ap.
|
|
|
|
|
|
|
| 10 |
ap.add_argument('--base',default='Qwen/Qwen-Image-2.1')
|
| 11 |
ap.add_argument('--quant',default=str(Path(__file__).parent/'transformer'))
|
| 12 |
ap.add_argument('--width',type=int,default=2048)
|
|
@@ -15,17 +18,37 @@ def main():
|
|
| 15 |
ap.add_argument('--seed',type=int,default=42)
|
| 16 |
ap.add_argument('--output',default='output.png')
|
| 17 |
ap.add_argument('--image')
|
|
|
|
|
|
|
| 18 |
args=ap.parse_args()
|
| 19 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 20 |
kw={}
|
| 21 |
if args.image:
|
| 22 |
from PIL import Image
|
| 23 |
kw['image']=Image.open(args.image)
|
| 24 |
-
|
| 25 |
-
|
| 26 |
-
|
| 27 |
-
|
| 28 |
-
|
| 29 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 30 |
|
| 31 |
if __name__=='__main__':main()
|
|
|
|
| 1 |
"""Generate with Image 2.1 Calibrated FP8. Built with Qwen."""
|
| 2 |
+
import argparse, json, time
|
| 3 |
from pathlib import Path
|
| 4 |
import torch
|
| 5 |
from fp8_runtime import load_pipeline
|
| 6 |
|
| 7 |
+
@torch.inference_mode()
|
| 8 |
def main():
|
| 9 |
ap=argparse.ArgumentParser()
|
| 10 |
+
inputs=ap.add_mutually_exclusive_group(required=True)
|
| 11 |
+
inputs.add_argument('--prompt')
|
| 12 |
+
inputs.add_argument('--prompts-json', help='JSON list of prompt strings; one loaded pipeline serves the whole list')
|
| 13 |
ap.add_argument('--base',default='Qwen/Qwen-Image-2.1')
|
| 14 |
ap.add_argument('--quant',default=str(Path(__file__).parent/'transformer'))
|
| 15 |
ap.add_argument('--width',type=int,default=2048)
|
|
|
|
| 18 |
ap.add_argument('--seed',type=int,default=42)
|
| 19 |
ap.add_argument('--output',default='output.png')
|
| 20 |
ap.add_argument('--image')
|
| 21 |
+
ap.add_argument('--eager', action='store_true', help='Use the original uncompiled runtime')
|
| 22 |
+
ap.add_argument('--warmup', action='store_true', help='Run a full untimed warmup; report its cost separately')
|
| 23 |
args=ap.parse_args()
|
| 24 |
+
prompts=json.loads(Path(args.prompts_json).read_text()) if args.prompts_json else [args.prompt]
|
| 25 |
+
if not isinstance(prompts,list) or not prompts or not all(isinstance(p,str) and p for p in prompts):
|
| 26 |
+
ap.error('prompts-json must contain a nonempty list of prompt strings')
|
| 27 |
+
start=time.perf_counter();pipe=load_pipeline(args.base,args.quant)
|
| 28 |
+
if not args.eager:
|
| 29 |
+
from acceleration import accelerate_pipeline
|
| 30 |
+
accelerate_pipeline(pipe)
|
| 31 |
+
torch.cuda.synchronize();load_seconds=time.perf_counter()-start
|
| 32 |
kw={}
|
| 33 |
if args.image:
|
| 34 |
from PIL import Image
|
| 35 |
kw['image']=Image.open(args.image)
|
| 36 |
+
def run(prompt,seed):
|
| 37 |
+
return pipe(prompt=prompt,width=args.width,height=args.height,num_inference_steps=args.steps,
|
| 38 |
+
generator=torch.Generator('cuda').manual_seed(seed),**kw).images[0]
|
| 39 |
+
warmup_seconds=0
|
| 40 |
+
if args.warmup:
|
| 41 |
+
start=time.perf_counter();run(prompts[0],args.seed);torch.cuda.synchronize()
|
| 42 |
+
warmup_seconds=time.perf_counter()-start
|
| 43 |
+
path=Path(args.output);path.parent.mkdir(parents=True,exist_ok=True)
|
| 44 |
+
for i,prompt in enumerate(prompts):
|
| 45 |
+
torch.cuda.synchronize();start=time.perf_counter();result=run(prompt,args.seed+i)
|
| 46 |
+
torch.cuda.synchronize();seconds=time.perf_counter()-start
|
| 47 |
+
destination=path if len(prompts)==1 else path.with_name(f'{path.stem}-{i:03d}{path.suffix or ".png"}')
|
| 48 |
+
result.save(destination)
|
| 49 |
+
print(json.dumps({'seconds':seconds,'output':str(destination),'width':result.width,'height':result.height,
|
| 50 |
+
'steps':args.steps,'eager':args.eager,'load_seconds':load_seconds,
|
| 51 |
+
'warmup_seconds':warmup_seconds,
|
| 52 |
+
'timing_note':'Generation includes any compilation on this call; excludes model load and PNG writing.'}),flush=True)
|
| 53 |
|
| 54 |
if __name__=='__main__':main()
|
manifest.json
CHANGED
|
@@ -12,8 +12,13 @@
|
|
| 12 |
},
|
| 13 |
{
|
| 14 |
"path": "README.md",
|
| 15 |
-
"bytes":
|
| 16 |
-
"sha256": "
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 17 |
},
|
| 18 |
{
|
| 19 |
"path": "calibrate.py",
|
|
@@ -145,6 +150,16 @@
|
|
| 145 |
"bytes": 1525508,
|
| 146 |
"sha256": "49221b8292e06602e20a146c2646d8c31aa7e662f64b19bfa546de652fa6e60c"
|
| 147 |
},
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 148 |
{
|
| 149 |
"path": "demo/realtime-30s.mp4",
|
| 150 |
"bytes": 1529663,
|
|
@@ -162,14 +177,264 @@
|
|
| 162 |
},
|
| 163 |
{
|
| 164 |
"path": "generate.py",
|
| 165 |
-
"bytes":
|
| 166 |
-
"sha256": "
|
| 167 |
},
|
| 168 |
{
|
| 169 |
"path": "metrics.py",
|
| 170 |
"bytes": 3076,
|
| 171 |
"sha256": "30a92e4dfc5159e68219422c33c7f75b7c8c88c558a42f931f6c86f87d09633b"
|
| 172 |
},
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 173 |
{
|
| 174 |
"path": "prepare_release.py",
|
| 175 |
"bytes": 9849,
|
|
@@ -391,5 +656,5 @@
|
|
| 391 |
"sha256": "5530a9c9111e68454b1a09f985337b3a31217e2b6f2f74401ed4d1c7f81b44c2"
|
| 392 |
}
|
| 393 |
],
|
| 394 |
-
"total_bytes":
|
| 395 |
}
|
|
|
|
| 12 |
},
|
| 13 |
{
|
| 14 |
"path": "README.md",
|
| 15 |
+
"bytes": 9193,
|
| 16 |
+
"sha256": "675f85d35555814eda4611661c0c6087590d7b23e962e5f791e5d91ee9a46c5f"
|
| 17 |
+
},
|
| 18 |
+
{
|
| 19 |
+
"path": "acceleration.py",
|
| 20 |
+
"bytes": 4131,
|
| 21 |
+
"sha256": "733f8071b9d577df7cf3f3c0f60e039822e85fdbc8f4d0446f6edf4b8cde2b3d"
|
| 22 |
},
|
| 23 |
{
|
| 24 |
"path": "calibrate.py",
|
|
|
|
| 150 |
"bytes": 1525508,
|
| 151 |
"sha256": "49221b8292e06602e20a146c2646d8c31aa7e662f64b19bfa546de652fa6e60c"
|
| 152 |
},
|
| 153 |
+
{
|
| 154 |
+
"path": "demo/optimized-capture_receipt.json",
|
| 155 |
+
"bytes": 131530,
|
| 156 |
+
"sha256": "8ce78a2b4cd5d6e7919ac83995258b56abedec844e87b82759d1210cf9188522"
|
| 157 |
+
},
|
| 158 |
+
{
|
| 159 |
+
"path": "demo/optimized-realtime-30s.mp4",
|
| 160 |
+
"bytes": 1753321,
|
| 161 |
+
"sha256": "3271da65a4797dda30419d7857612fb05873497f2c50b18056c96fde92fb9457"
|
| 162 |
+
},
|
| 163 |
{
|
| 164 |
"path": "demo/realtime-30s.mp4",
|
| 165 |
"bytes": 1529663,
|
|
|
|
| 177 |
},
|
| 178 |
{
|
| 179 |
"path": "generate.py",
|
| 180 |
+
"bytes": 3003,
|
| 181 |
+
"sha256": "fa5c0b0f25f5b36e0449f04f9a5b014097433023f791309b1a6a0b95538f97e9"
|
| 182 |
},
|
| 183 |
{
|
| 184 |
"path": "metrics.py",
|
| 185 |
"bytes": 3076,
|
| 186 |
"sha256": "30a92e4dfc5159e68219422c33c7f75b7c8c88c558a42f931f6c86f87d09633b"
|
| 187 |
},
|
| 188 |
+
{
|
| 189 |
+
"path": "optimization/REPORT.md",
|
| 190 |
+
"bytes": 4520,
|
| 191 |
+
"sha256": "bc07ca5d94b402cfa8c85d9eceac750dfad1b064654c783b2dcc269f48bfadff"
|
| 192 |
+
},
|
| 193 |
+
{
|
| 194 |
+
"path": "optimization/baseline-benchmark.json",
|
| 195 |
+
"bytes": 915,
|
| 196 |
+
"sha256": "6a3b90e7fcba23deacbdb86e4f5703221b3dfce5e1800e9e1f0be4903520b2e1"
|
| 197 |
+
},
|
| 198 |
+
{
|
| 199 |
+
"path": "optimization/benchmark.json",
|
| 200 |
+
"bytes": 1022,
|
| 201 |
+
"sha256": "ab5b691da6e886bbf82a7c96928466b51d05847d03f98ec0d4c708a21212244a"
|
| 202 |
+
},
|
| 203 |
+
{
|
| 204 |
+
"path": "optimization/comparisons/00.jpg",
|
| 205 |
+
"bytes": 175133,
|
| 206 |
+
"sha256": "88a1d96deadd111e6e61e9b8faa0563e6189f511e39eecf0439bc76c794ac1cf"
|
| 207 |
+
},
|
| 208 |
+
{
|
| 209 |
+
"path": "optimization/comparisons/01.jpg",
|
| 210 |
+
"bytes": 181067,
|
| 211 |
+
"sha256": "ea9eb200c963a44b4ab7ef3dbf4400456d577a332d8d1d86d574598ca74ac297"
|
| 212 |
+
},
|
| 213 |
+
{
|
| 214 |
+
"path": "optimization/comparisons/02.jpg",
|
| 215 |
+
"bytes": 180763,
|
| 216 |
+
"sha256": "dce907958df0637061437c6e2eca2d8b81d7ef034aff35ae3d3524926eda7591"
|
| 217 |
+
},
|
| 218 |
+
{
|
| 219 |
+
"path": "optimization/comparisons/03.jpg",
|
| 220 |
+
"bytes": 143324,
|
| 221 |
+
"sha256": "81db79bdafec7be1a96a4f44dfba26d03b32dc5c46946e20c0dfb89129d958cf"
|
| 222 |
+
},
|
| 223 |
+
{
|
| 224 |
+
"path": "optimization/comparisons/04.jpg",
|
| 225 |
+
"bytes": 183027,
|
| 226 |
+
"sha256": "ac74b465ef17e6b4609542b88d88fdb8500b0947c7e76300e26446c5d20c22c2"
|
| 227 |
+
},
|
| 228 |
+
{
|
| 229 |
+
"path": "optimization/comparisons/05.jpg",
|
| 230 |
+
"bytes": 315841,
|
| 231 |
+
"sha256": "ca0902b94c869e4616290d5600e9533c5d3805babecda284b3f532a4b36c09c5"
|
| 232 |
+
},
|
| 233 |
+
{
|
| 234 |
+
"path": "optimization/comparisons/06.jpg",
|
| 235 |
+
"bytes": 103060,
|
| 236 |
+
"sha256": "281c7cd5a5cd8a89177c74a5a4c87a291f9328d5a157c9586dc8cbb250f73c84"
|
| 237 |
+
},
|
| 238 |
+
{
|
| 239 |
+
"path": "optimization/comparisons/07.jpg",
|
| 240 |
+
"bytes": 120711,
|
| 241 |
+
"sha256": "5730fabf13bc88997e29eb0962907d4e7550ca8fc2df6f4be7c1026ccc2cee85"
|
| 242 |
+
},
|
| 243 |
+
{
|
| 244 |
+
"path": "optimization/comparisons/08.jpg",
|
| 245 |
+
"bytes": 112707,
|
| 246 |
+
"sha256": "e8bccf4d01d631f87809e03c0cfd3ffba990be9943882d6299b0fc0759560115"
|
| 247 |
+
},
|
| 248 |
+
{
|
| 249 |
+
"path": "optimization/comparisons/09.jpg",
|
| 250 |
+
"bytes": 98850,
|
| 251 |
+
"sha256": "fabb5456e0653a01be53e54ee1fad9e23f951b935bd25db8fd4c0627301a5d3a"
|
| 252 |
+
},
|
| 253 |
+
{
|
| 254 |
+
"path": "optimization/comparisons/10.jpg",
|
| 255 |
+
"bytes": 60297,
|
| 256 |
+
"sha256": "c9b00fce315e13c5688f8d7f61ef79826fe0ad4a5ffd4d97e66ef09abc6d40e4"
|
| 257 |
+
},
|
| 258 |
+
{
|
| 259 |
+
"path": "optimization/comparisons/11.jpg",
|
| 260 |
+
"bytes": 265201,
|
| 261 |
+
"sha256": "c2ed073577c1013d134aba048527858ef85791b6fe0ee17f6c9d4cd8d4125900"
|
| 262 |
+
},
|
| 263 |
+
{
|
| 264 |
+
"path": "optimization/comparisons/12.jpg",
|
| 265 |
+
"bytes": 276087,
|
| 266 |
+
"sha256": "715a235e24d3cda41c450d0bdeb70b9727a5e1b247ab65c9f49b80ee7f3d08da"
|
| 267 |
+
},
|
| 268 |
+
{
|
| 269 |
+
"path": "optimization/comparisons/13.jpg",
|
| 270 |
+
"bytes": 86206,
|
| 271 |
+
"sha256": "d29c588a172de01dfcd63b875e85c81cf3400948d148a44a4a4f6f0adb30ca67"
|
| 272 |
+
},
|
| 273 |
+
{
|
| 274 |
+
"path": "optimization/comparisons/14.jpg",
|
| 275 |
+
"bytes": 191026,
|
| 276 |
+
"sha256": "7653a0daaad9c6ce0007b3deefe81285ad8cb5b96041214e94b9c78be8a23821"
|
| 277 |
+
},
|
| 278 |
+
{
|
| 279 |
+
"path": "optimization/comparisons/15.jpg",
|
| 280 |
+
"bytes": 201338,
|
| 281 |
+
"sha256": "f9921556d46589c2f468185011cc8923d25d86d559e3e68e28302b0a6572a3f6"
|
| 282 |
+
},
|
| 283 |
+
{
|
| 284 |
+
"path": "optimization/comparisons/edit-0.jpg",
|
| 285 |
+
"bytes": 230651,
|
| 286 |
+
"sha256": "b37d732075b4d1027581d91eb1f6e317e147a412f364ac6349dbc30a0313f3ae"
|
| 287 |
+
},
|
| 288 |
+
{
|
| 289 |
+
"path": "optimization/comparisons/edit-1.jpg",
|
| 290 |
+
"bytes": 137443,
|
| 291 |
+
"sha256": "9b65249cf9578d15693497f6dc59fc2c3163901258c568ca47e31cf803d814d9"
|
| 292 |
+
},
|
| 293 |
+
{
|
| 294 |
+
"path": "optimization/denoising-trace.json",
|
| 295 |
+
"bytes": 3769855,
|
| 296 |
+
"sha256": "3654ad7b284d6591c84bc7e261f6dba3cfdd0c9fee491cba53e8dcb0595d3e9d"
|
| 297 |
+
},
|
| 298 |
+
{
|
| 299 |
+
"path": "optimization/kernel_evidence.json",
|
| 300 |
+
"bytes": 25421,
|
| 301 |
+
"sha256": "0ba67d4ebc96ba34188b709fbcb5ccf2ce797ab1a9a5e20755027c2b55a13ca5"
|
| 302 |
+
},
|
| 303 |
+
{
|
| 304 |
+
"path": "optimization/qa-review.json",
|
| 305 |
+
"bytes": 1741,
|
| 306 |
+
"sha256": "78d764090f9332ac2f87027d3aab43a056e50b9c047bf0a184a825d4ee68330b"
|
| 307 |
+
},
|
| 308 |
+
{
|
| 309 |
+
"path": "optimization/quality-vs-bf16.json",
|
| 310 |
+
"bytes": 5366,
|
| 311 |
+
"sha256": "17fb14656485ba2fae1257dd26a42eab392d05333d815fcffd53a164286552c6"
|
| 312 |
+
},
|
| 313 |
+
{
|
| 314 |
+
"path": "optimization/quality-vs-fp8.json",
|
| 315 |
+
"bytes": 5377,
|
| 316 |
+
"sha256": "8d7eeaa46947f96807b8af9d9dee54703fe4bc556fc1a87614f7b04d34ff60d6"
|
| 317 |
+
},
|
| 318 |
+
{
|
| 319 |
+
"path": "optimization/samples/00.png",
|
| 320 |
+
"bytes": 7114893,
|
| 321 |
+
"sha256": "3c2a6ccae55ac24274c4837266fc2f3730860280435cd119d66b0830b74b8f31"
|
| 322 |
+
},
|
| 323 |
+
{
|
| 324 |
+
"path": "optimization/samples/01.png",
|
| 325 |
+
"bytes": 1841389,
|
| 326 |
+
"sha256": "bba7ebb9972b517dc7cb732a729ee27d4432dd2476876acb281b0cafde02d7ca"
|
| 327 |
+
},
|
| 328 |
+
{
|
| 329 |
+
"path": "optimization/samples/02.png",
|
| 330 |
+
"bytes": 1759707,
|
| 331 |
+
"sha256": "95ee80540a4357536c00e9c06cbcae7810340b99626e6e1b91746c48a78fd6a7"
|
| 332 |
+
},
|
| 333 |
+
{
|
| 334 |
+
"path": "optimization/samples/03.png",
|
| 335 |
+
"bytes": 1643411,
|
| 336 |
+
"sha256": "5efef19fad659b46794e4beda47216cc08cf9fa3eb4d3bc53cf7fc2d8a10c17c"
|
| 337 |
+
},
|
| 338 |
+
{
|
| 339 |
+
"path": "optimization/samples/04.png",
|
| 340 |
+
"bytes": 6445668,
|
| 341 |
+
"sha256": "62df04f463aa5a483b1e05aaff43662f9469b7a1d3711aaf0a162bca9be17d39"
|
| 342 |
+
},
|
| 343 |
+
{
|
| 344 |
+
"path": "optimization/samples/05.png",
|
| 345 |
+
"bytes": 2530758,
|
| 346 |
+
"sha256": "6b4643e1a72f7a5410608dd1674f7bfb6d01aefb3d810cb82c7386b4ab6d3cdc"
|
| 347 |
+
},
|
| 348 |
+
{
|
| 349 |
+
"path": "optimization/samples/06.png",
|
| 350 |
+
"bytes": 1364614,
|
| 351 |
+
"sha256": "f7a12a23429639f9b36b21499bf63b012eae750cfa73eb36c9824104f656acc4"
|
| 352 |
+
},
|
| 353 |
+
{
|
| 354 |
+
"path": "optimization/samples/07.png",
|
| 355 |
+
"bytes": 1474237,
|
| 356 |
+
"sha256": "d647b4ef888654980a06937cc05bfc6ce1dd96efa3811eff213217ca4714c4e2"
|
| 357 |
+
},
|
| 358 |
+
{
|
| 359 |
+
"path": "optimization/samples/08.png",
|
| 360 |
+
"bytes": 3659644,
|
| 361 |
+
"sha256": "8b014521e112db7238f332905bff3b8159a9e961889366fdf5681845e60bcf72"
|
| 362 |
+
},
|
| 363 |
+
{
|
| 364 |
+
"path": "optimization/samples/09.png",
|
| 365 |
+
"bytes": 1196128,
|
| 366 |
+
"sha256": "1053bbf49bd147e6c6ab8f58567cc51d536e72089ffa94886b5bcf5f7723f8ef"
|
| 367 |
+
},
|
| 368 |
+
{
|
| 369 |
+
"path": "optimization/samples/10.png",
|
| 370 |
+
"bytes": 548987,
|
| 371 |
+
"sha256": "ad2cf4fb36af7474b51c432afe8664458a15d4f5a2e3bee63073806ba89fe877"
|
| 372 |
+
},
|
| 373 |
+
{
|
| 374 |
+
"path": "optimization/samples/11.png",
|
| 375 |
+
"bytes": 2247679,
|
| 376 |
+
"sha256": "b5544930a4022592a2b977c25a265cdd73cd1f6eaec1d540c1fe9d0af7fd12ac"
|
| 377 |
+
},
|
| 378 |
+
{
|
| 379 |
+
"path": "optimization/samples/12.png",
|
| 380 |
+
"bytes": 3162596,
|
| 381 |
+
"sha256": "24a122bc9735495fc3dfdaea28f39dcec137c98cc3bf97e90663685b1cff7066"
|
| 382 |
+
},
|
| 383 |
+
{
|
| 384 |
+
"path": "optimization/samples/13.png",
|
| 385 |
+
"bytes": 1432254,
|
| 386 |
+
"sha256": "f67fe261fea1044c0372782c3e34bcda9760de289d02fae16468c9f5958b9192"
|
| 387 |
+
},
|
| 388 |
+
{
|
| 389 |
+
"path": "optimization/samples/14.png",
|
| 390 |
+
"bytes": 1794850,
|
| 391 |
+
"sha256": "867802d5477883ada08eda5e3485387ba2d95f9112a5e069dd5468d0abd98e78"
|
| 392 |
+
},
|
| 393 |
+
{
|
| 394 |
+
"path": "optimization/samples/15.png",
|
| 395 |
+
"bytes": 2013263,
|
| 396 |
+
"sha256": "06d10de93cc1a112f1be0a9935973fa525b8192187f545e9925c8354ec59f3d0"
|
| 397 |
+
},
|
| 398 |
+
{
|
| 399 |
+
"path": "optimization/samples/edit-0.png",
|
| 400 |
+
"bytes": 1979896,
|
| 401 |
+
"sha256": "0782c4024c8461e1b41bc51ee8b1d262424cf2f9478514f10d9155e533aaad53"
|
| 402 |
+
},
|
| 403 |
+
{
|
| 404 |
+
"path": "optimization/samples/edit-1.png",
|
| 405 |
+
"bytes": 1569214,
|
| 406 |
+
"sha256": "5b283cc81788a506374b2e4db6aee295cb145ef96c670028c89ebc551caf854b"
|
| 407 |
+
},
|
| 408 |
+
{
|
| 409 |
+
"path": "optimization/source/acceleration.py",
|
| 410 |
+
"bytes": 4131,
|
| 411 |
+
"sha256": "733f8071b9d577df7cf3f3c0f60e039822e85fdbc8f4d0446f6edf4b8cde2b3d"
|
| 412 |
+
},
|
| 413 |
+
{
|
| 414 |
+
"path": "optimization/source/benchmark_final.py",
|
| 415 |
+
"bytes": 3464,
|
| 416 |
+
"sha256": "f7943703bfa4895635bfd67da97a83a49ee35cd6a9aa92ad5f19719038b6ad98"
|
| 417 |
+
},
|
| 418 |
+
{
|
| 419 |
+
"path": "optimization/source/quality_metrics.py",
|
| 420 |
+
"bytes": 3287,
|
| 421 |
+
"sha256": "f02833d14e59fefb434b8331cdc92e50300028354a1a35959377485803d16763"
|
| 422 |
+
},
|
| 423 |
+
{
|
| 424 |
+
"path": "optimization/source/validate.py",
|
| 425 |
+
"bytes": 2494,
|
| 426 |
+
"sha256": "b5d7df99f713c814028fbb1a7652e99f29eb373dd050abc5d394f0c5855cf8dc"
|
| 427 |
+
},
|
| 428 |
+
{
|
| 429 |
+
"path": "optimization/source/video_optimized.py",
|
| 430 |
+
"bytes": 3910,
|
| 431 |
+
"sha256": "cde9dda4ec915125a7b6a80face7a32d02117da674f2d691e83d0b790005cf15"
|
| 432 |
+
},
|
| 433 |
+
{
|
| 434 |
+
"path": "optimization/video-verification.json",
|
| 435 |
+
"bytes": 1860,
|
| 436 |
+
"sha256": "88052dfae9c6f7ac749791b38646b738a142b5eeef0a7dfc9992b6d3dd5b6e84"
|
| 437 |
+
},
|
| 438 |
{
|
| 439 |
"path": "prepare_release.py",
|
| 440 |
"bytes": 9849,
|
|
|
|
| 656 |
"sha256": "5530a9c9111e68454b1a09f985337b3a31217e2b6f2f74401ed4d1c7f81b44c2"
|
| 657 |
}
|
| 658 |
],
|
| 659 |
+
"total_bytes": 7382689916
|
| 660 |
}
|
optimization/REPORT.md
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Runtime optimization with unchanged denoising precision
|
| 2 |
+
|
| 3 |
+
## Full-compute runtime update
|
| 4 |
+
|
| 5 |
+
The default `generate.py` now uses **decode-only dynamic compilation and kernel fusion**, keeping all 40 denoising steps, original calibrated FP8 weights/scales, FP32 GEMM accumulation, and native BF16 attention. No step reuse, distilled adapter, further attention quantization, or reduced resolution is enabled. `--eager` selects the original runtime.
|
| 6 |
+
|
| 7 |
+
| Resolution | Original FP8, fresh control | Optimized FP8 | Speedup |
|
| 8 |
+
|---|---:|---:|---:|
|
| 9 |
+
| 1024×1024 | 6.922 s | 5.932 s | 1.167× |
|
| 10 |
+
| 2048×2048 | 44.101 s | 37.289 s | 1.183× |
|
| 11 |
+
|
| 12 |
+
Same RTX PRO 6000, batch 1, 40 steps, CFG 1, prefix KV cache. CUDA-synchronized end-to-end generation includes encoding and VAE. Warmup/model loading/PNG writes excluded. Original control has 2 measured repeats per size; optimized has 5 at 1024 and 3 at 2048. [Raw measurements](benchmark.json) include first-use warmup and load costs. First-use compilation costs extra time. `--warmup` reports that cost separately; `--prompts-json prompts.json` processes a list of prompt strings in one loaded pipeline.
|
| 13 |
+
|
| 14 |
+
All 18 paired checks were visually reviewed, including primary English/Chinese titles, portraits, transparency and edits. No obvious general visual-quality loss was observed in this finite set. **Outputs are not bit-identical**: fine textures, decorative typography and some pottery positioning change with GPU reduction rounding. Mean LPIPS versus original FP8=0.014105, mean SSIM=0.985841; worst LPIPS=0.104154 (pottery). Relative to BF16, mean LPIPS=0.035437, compared with 0.034470 for the original FP8. These are fidelity checks, not a guarantee for every prompt. [Comparisons](comparisons/) · [Detailed report](REPORT.md).
|
| 15 |
+
|
| 16 |
+
[Updated 30-second real-time image-only video](../demo/optimized-realtime-30s.mp4). One completed warmup image is shown at time 0, then each new image appears when actual generation finishes. All waits remain at 1× speed. [Frame/request timestamps](../demo/optimized-capture_receipt.json).
|
| 17 |
+
|
| 18 |
+
|
| 19 |
+
## Implementation
|
| 20 |
+
|
| 21 |
+
`acceleration.py` compiles the 32 repeated transformer blocks for cached-prefix decode, while keeping prefill in the upstream eager path. Dynamic lengths avoid per-prompt recompilation after warming relevant resolution shapes. Real-valued rotary arithmetic is fused with surrounding operations. Cached decode contains only target tokens, allowing direct per-sample modulation broadcasting instead of materializing a target/prefix mask across every token. Inductor `emulate_precision_casts=True` preserves intermediate BF16 rounding boundaries. Floating-point GPU reduction ordering can still differ.
|
| 22 |
+
|
| 23 |
+
Weights and calibration were not changed. Actual denoising-step profiling still observes 224 native CUTLASS SM120 E4M3 weight GEMMs and 32 native BF16 FlashAttention kernels. Exactly 40 transformer calls occur for 40 steps. See [kernel evidence](kernel_evidence.json) and [trace](denoising-trace.json).
|
| 24 |
+
|
| 25 |
+
## Investigated and not adopted
|
| 26 |
+
|
| 27 |
+
Fast FP8 accumulation, cuDNN attention, FlashAttention4 and a 24-configuration Triton FP8 GEMM probe did not offer a meaningful applicable improvement. SageAttention 2 and conservative first-block residual reuse produced larger gains, but introduce additional approximation; both were excluded following the explicit quality requirement. Neither is required or enabled by this release.
|
| 28 |
+
|
| 29 |
+
## Quality review
|
| 30 |
+
|
| 31 |
+
No obvious general visual quality loss was observed in all 18 paired images, including portraits, wildlife, macro, food, architecture, paintings, glass, primary English and Chinese titles, transparency, crafts, coastline, flowers, human hands, and two edits. Outputs are not bit-identical: potter arm and clay positioning, snow leopard coat and background details, decorative typography, and some fine textures differ. The pottery sample has the largest LPIPS difference (0.10415) and remains visually plausible. This finite review cannot guarantee every prompt.
|
| 32 |
+
|
| 33 |
+
The original release remains immutable at [f642d1f49f031f550655d07ca4569ff729ba9cfd](https://huggingface.co/ProCreations/Image-2.1-Calibrated-FP8/tree/f642d1f49f031f550655d07ca4569ff729ba9cfd). `--eager` is available for the original execution path. Tested Torch 2.14.0+cu130, Triton 3.8.0, Transformers 5.17.0, Diffusers 80c7ed262aeffbeb43ef13ae04baeb9b84515a69, RTX PRO 6000 SM120. Research license unchanged.
|
| 34 |
+
|
| 35 |
+
The source scripts retain explicit workstation layout assumptions; adjust input/output paths for another installation.
|
optimization/baseline-benchmark.json
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"load_seconds": 2.426567278977018,
|
| 3 |
+
"torch": "2.14.0+cu130",
|
| 4 |
+
"gpu": "NVIDIA RTX PRO 6000 Blackwell Workstation Edition",
|
| 5 |
+
"fp8_linears": 224,
|
| 6 |
+
"steps": 40,
|
| 7 |
+
"cfg": 1,
|
| 8 |
+
"extra_quantization": false,
|
| 9 |
+
"approximate_cache": false,
|
| 10 |
+
"timing": {
|
| 11 |
+
"1024": {
|
| 12 |
+
"seconds": [
|
| 13 |
+
6.9087469020159915,
|
| 14 |
+
6.936133738025092
|
| 15 |
+
],
|
| 16 |
+
"mean": 6.9224403200205415,
|
| 17 |
+
"warmup_seconds": 8.378067673009355,
|
| 18 |
+
"peak_gb": 32.590829568
|
| 19 |
+
},
|
| 20 |
+
"2048": {
|
| 21 |
+
"seconds": [
|
| 22 |
+
44.079785163048655,
|
| 23 |
+
44.1227304089698
|
| 24 |
+
],
|
| 25 |
+
"mean": 44.10125778600923,
|
| 26 |
+
"warmup_seconds": 43.912163228029385,
|
| 27 |
+
"peak_gb": 53.722432512
|
| 28 |
+
}
|
| 29 |
+
},
|
| 30 |
+
"protocol": "CUDA synchronized; batch1; full40steps; includes encoder, denoising and VAE; excludes model load, resolution warmup and file writes. Weights and prefixKVcache unchanged. Compiled mode emulates intermediate precision casts."
|
| 31 |
+
}
|
optimization/benchmark.json
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"load_seconds": 2.426407459017355,
|
| 3 |
+
"torch": "2.14.0+cu130",
|
| 4 |
+
"gpu": "NVIDIA RTX PRO 6000 Blackwell Workstation Edition",
|
| 5 |
+
"fp8_linears": 224,
|
| 6 |
+
"steps": 40,
|
| 7 |
+
"cfg": 1,
|
| 8 |
+
"extra_quantization": false,
|
| 9 |
+
"approximate_cache": false,
|
| 10 |
+
"timing": {
|
| 11 |
+
"1024": {
|
| 12 |
+
"seconds": [
|
| 13 |
+
5.903008527995553,
|
| 14 |
+
5.9249284460092895,
|
| 15 |
+
5.935668507008813,
|
| 16 |
+
5.946647913951892,
|
| 17 |
+
5.951301124005113
|
| 18 |
+
],
|
| 19 |
+
"mean": 5.932310903794132,
|
| 20 |
+
"warmup_seconds": 11.555044600041583,
|
| 21 |
+
"peak_gb": 32.590829568
|
| 22 |
+
},
|
| 23 |
+
"2048": {
|
| 24 |
+
"seconds": [
|
| 25 |
+
37.25538438500371,
|
| 26 |
+
37.301657135016285,
|
| 27 |
+
37.31008757499512
|
| 28 |
+
],
|
| 29 |
+
"mean": 37.2890430316717,
|
| 30 |
+
"warmup_seconds": 39.20627289195545,
|
| 31 |
+
"peak_gb": 53.722432512
|
| 32 |
+
}
|
| 33 |
+
},
|
| 34 |
+
"protocol": "CUDA synchronized; batch1; full40steps; includes encoder, denoising and VAE; excludes model load, resolution warmup and file writes. Weights and prefixKVcache unchanged. Compiled mode emulates intermediate precision casts."
|
| 35 |
+
}
|
optimization/comparisons/00.jpg
ADDED
|
Git LFS Details
|
optimization/comparisons/01.jpg
ADDED
|
Git LFS Details
|
optimization/comparisons/02.jpg
ADDED
|
Git LFS Details
|
optimization/comparisons/03.jpg
ADDED
|
Git LFS Details
|
optimization/comparisons/04.jpg
ADDED
|
Git LFS Details
|
optimization/comparisons/05.jpg
ADDED
|
Git LFS Details
|
optimization/comparisons/06.jpg
ADDED
|
Git LFS Details
|
optimization/comparisons/07.jpg
ADDED
|
Git LFS Details
|
optimization/comparisons/08.jpg
ADDED
|
Git LFS Details
|
optimization/comparisons/09.jpg
ADDED
|
optimization/comparisons/10.jpg
ADDED
|
optimization/comparisons/11.jpg
ADDED
|
Git LFS Details
|
optimization/comparisons/12.jpg
ADDED
|
Git LFS Details
|
optimization/comparisons/13.jpg
ADDED
|
optimization/comparisons/14.jpg
ADDED
|
Git LFS Details
|
optimization/comparisons/15.jpg
ADDED
|
Git LFS Details
|
optimization/comparisons/edit-0.jpg
ADDED
|
Git LFS Details
|
optimization/comparisons/edit-1.jpg
ADDED
|
Git LFS Details
|
optimization/denoising-trace.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
optimization/kernel_evidence.json
ADDED
|
@@ -0,0 +1,75 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"fp8_sm120_launches": 224,
|
| 3 |
+
"native_bf16_flash_attention": 32,
|
| 4 |
+
"all_kernels": {
|
| 5 |
+
"void cutlass::Kernel2<cutlass_80_tensorop_bf16_s16816gemm_relu_bf16_128x256_32x3_tt_align8>(cutlass_80_tensorop_bf16_s16816gemm_relu_bf16_128x256_32x3_tt_align8::Params)": 1,
|
| 6 |
+
"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda()#3}::operator()() const::{lambda()#7}::operator()() const::{lambda(float)#1}, std::array<char*, 2ul>, 4, TrivialOffsetCalculator<1, unsigned int>, TrivialOffsetCalculator<1, unsigned int>, at::native::memory::LoadWithCast<1>, at::native::memory::StoreWithCast<1> >(int, at::native::direct_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda()#3}::operator()() const::{lambda()#7}::operator()() const::{lambda(float)#1}, std::array<char*, 2ul>, TrivialOffsetCalculator<1, unsigned int>, TrivialOffsetCalculator<1, unsigned int>, at::native::memory::LoadWithCast<1>, at::native::memory::StoreWithCast<1>)": 3,
|
| 7 |
+
"void at::native::vectorized_elementwise_kernel<4, at::native::(anonymous namespace)::pow_tensor_scalar_kernel_impl<float, float>(at::TensorIteratorBase&, float)::{lambda(float)#1}, std::array<char*, 2ul>, false>(int, at::native::(anonymous namespace)::pow_tensor_scalar_kernel_impl<float, float>(at::TensorIteratorBase&, float)::{lambda(float)#1}, std::array<char*, 2ul>)": 1,
|
| 8 |
+
"void at::native::reduce_kernel<512, 1, at::native::ReduceOp<float, at::native::MeanOps<float, float, float, float>, unsigned int, float, 4, 4> >(at::native::ReduceOp<float, at::native::MeanOps<float, float, float, float>, unsigned int, float, 4, 4>)": 1,
|
| 9 |
+
"void at::native::vectorized_elementwise_kernel<4, at::native::CUDAFunctorOnSelf_add<float>, std::array<char*, 2ul>, false>(int, at::native::CUDAFunctorOnSelf_add<float>, std::array<char*, 2ul>)": 2,
|
| 10 |
+
"void at::native::vectorized_elementwise_kernel<4, at::native::rsqrt_kernel_cuda(at::TensorIteratorBase&)::{lambda()#2}::operator()() const::{lambda()#2}::operator()() const::{lambda(float)#1}, std::array<char*, 2ul>, false>(int, at::native::rsqrt_kernel_cuda(at::TensorIteratorBase&)::{lambda()#2}::operator()() const::{lambda()#2}::operator()() const::{lambda(float)#1}, std::array<char*, 2ul>)": 1,
|
| 11 |
+
"void at::native::elementwise_kernel<128, 2, at::native::gpu_kernel_impl_nocast<at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float> > >(at::TensorIteratorBase&, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float> > const&)::{lambda(int)#1}>(int, at::native::gpu_kernel_impl_nocast<at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float> > >(at::TensorIteratorBase&, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float> > const&)::{lambda(int)#1})": 3,
|
| 12 |
+
"void at::native::vectorized_elementwise_kernel<4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda(float)#1}, std::array<char*, 2ul>, false>(int, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda(float)#1}, std::array<char*, 2ul>)": 2,
|
| 13 |
+
"void cutlass::Kernel2<cutlass_80_tensorop_bf16_s16816gemm_relu_bf16_128x64_32x6_tn_align8>(cutlass_80_tensorop_bf16_s16816gemm_relu_bf16_128x64_32x6_tn_align8::Params)": 2,
|
| 14 |
+
"void cublasLt::splitKreduce_kernel<32, 16, int, __nv_bfloat16, __nv_bfloat16, float, __nv_bfloat16, false, __nv_bfloat16, __nv_bfloat16, __nv_bfloat16, true, false, false, false>(cublasLt::cublasSplitKParams<float>, __nv_bfloat16 const*, __nv_bfloat16 const*, __nv_bfloat16*, __nv_bfloat16*, float const*, float const*, __nv_bfloat16 const*, __nv_bfloat16 const*, __nv_bfloat16*, void*, long, float*, int*, float*, float*, float const*, float const*, float const*, float const*, float const*)": 3,
|
| 15 |
+
"void at::native::vectorized_elementwise_kernel<4, at::native::GeluCUDAKernelImpl(at::TensorIteratorBase&, at::native::GeluType)::{lambda()#1}::operator()() const::{lambda()#4}::operator()() const::{lambda(c10::BFloat16)#1}, std::array<char*, 2ul>, false>(int, at::native::GeluCUDAKernelImpl(at::TensorIteratorBase&, at::native::GeluType)::{lambda()#1}::operator()() const::{lambda()#4}::operator()() const::{lambda(c10::BFloat16)#1}, std::array<char*, 2ul>)": 1,
|
| 16 |
+
"void at::native::vectorized_elementwise_kernel<2, at::native::FillFunctor<long>, std::array<char*, 1ul>, false>(int, at::native::FillFunctor<long>, std::array<char*, 1ul>)": 3,
|
| 17 |
+
"void at::native::elementwise_kernel<128, 2, at::native::gpu_kernel_impl_nocast<at::native::(anonymous namespace)::where_kernel_impl(at::TensorIterator&)::{lambda()#1}::operator()() const::{lambda()#4}::operator()() const::{lambda(bool, unsigned long, unsigned long)#1}>(at::TensorIteratorBase&, at::native::(anonymous namespace)::where_kernel_impl(at::TensorIterator&)::{lambda()#1}::operator()() const::{lambda()#4}::operator()() const::{lambda(bool, unsigned long, unsigned long)#1} const&)::{lambda(int)#1}>(int, at::native::gpu_kernel_impl_nocast<at::native::(anonymous namespace)::where_kernel_impl(at::TensorIterator&)::{lambda()#1}::operator()() const::{lambda()#4}::operator()() const::{lambda(bool, unsigned long, unsigned long)#1}>(at::TensorIteratorBase&, at::native::(anonymous namespace)::where_kernel_impl(at::TensorIterator&)::{lambda()#1}::operator()() const::{lambda()#4}::operator()() const::{lambda(bool, unsigned long, unsigned long)#1} const&)::{lambda(int)#1})": 1,
|
| 18 |
+
"void at_cuda_detail::cub::detail::scan::DeviceScanInitKernel<at_cuda_detail::cub::ScanTileState<long, true> >(at_cuda_detail::cub::ScanTileState<long, true>, int)": 3,
|
| 19 |
+
"void at_cuda_detail::cub::detail::scan::DeviceScanKernel<at_cuda_detail::cub::detail::scan::policy_hub<long, long, long, unsigned int, std::plus<long> >::Policy1000, long const*, long*, at_cuda_detail::cub::ScanTileState<long, true>, std::plus<long>, at_cuda_detail::cub::NullType, unsigned int, long, false, at_cuda_detail::cub::NullType>(long const*, long*, at_cuda_detail::cub::ScanTileState<long, true>, int, std::plus<long>, at_cuda_detail::cub::NullType, unsigned int)": 3,
|
| 20 |
+
"Memcpy DtoH (Device -> Pinned)": 11,
|
| 21 |
+
"void at::native::vectorized_elementwise_kernel<4, at::native::compare_scalar_kernel<long>(at::TensorIteratorBase&, at::native::(anonymous namespace)::OpType, long)::{lambda(long)#1}, std::array<char*, 2ul>, false>(int, at::native::compare_scalar_kernel<long>(at::TensorIteratorBase&, at::native::(anonymous namespace)::OpType, long)::{lambda(long)#1}, std::array<char*, 2ul>)": 3,
|
| 22 |
+
"void at::native::reduce_kernel<512, 1, at::native::ReduceOp<bool, at::native::func_wrapper_t<bool, at::native::and_kernel_cuda(at::TensorIterator&)::{lambda()#1}::operator()() const::{lambda()#12}::operator()() const::{lambda(bool, bool)#1}>, unsigned int, bool, 4, 4> >(at::native::ReduceOp<bool, at::native::func_wrapper_t<bool, at::native::and_kernel_cuda(at::TensorIterator&)::{lambda()#1}::operator()() const::{lambda()#12}::operator()() const::{lambda(bool, bool)#1}>, unsigned int, bool, 4, 4>)": 2,
|
| 23 |
+
"void compute_cuda_kernel<long>(long const*, long const*, long*, long, long)": 3,
|
| 24 |
+
"void at::native::_scatter_gather_elementwise_kernel<128, 8, at::native::_cuda_scatter_gather_internal_kernel<false, at::native::OpaqueType<1>, long>::operator()<at::native::TensorAssign>(at::TensorIterator&, long, long, long, at::native::TensorAssign const&)::{lambda(int)#1}>(int, at::native::_cuda_scatter_gather_internal_kernel<false, at::native::OpaqueType<1>, long>::operator()<at::native::TensorAssign>(at::TensorIterator&, long, long, long, at::native::TensorAssign const&)::{lambda(int)#1})": 1,
|
| 25 |
+
"void at::native::vectorized_elementwise_kernel<4, at::native::FillFunctor<c10::BFloat16>, std::array<char*, 1ul>, false>(int, at::native::FillFunctor<c10::BFloat16>, std::array<char*, 1ul>)": 2,
|
| 26 |
+
"void at::native::(anonymous namespace)::CatArrayBatchedCopy_vectorized<at::native::(anonymous namespace)::OpaqueType<2u>, unsigned int, 2, 128, 1, 16, 8>(char*, at::native::(anonymous namespace)::CatArrInputTensorMetadata<at::native::(anonymous namespace)::OpaqueType<2u>, unsigned int, 128, 1>, at::native::(anonymous namespace)::TensorSizeStride<unsigned int, 4u>, int, unsigned int)": 1,
|
| 27 |
+
"void at::native::vectorized_gather_kernel<16, long>(char*, char*, long*, int, long, long, long, long, bool)": 4,
|
| 28 |
+
"void at_cuda_detail::cub::detail::reduce::DeviceReduceKernel<at_cuda_detail::cub::detail::reduce::policy_hub<int, unsigned long long, cuda::std::__4::plus<void> >::Policy1000, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::transform_iterator<at::native::(anonymous namespace)::NonZeroOp<bool>, bool const*, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default>, unsigned long long, cuda::std::__4::plus<void>, int, cuda::std::__4::__identity>(thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::transform_iterator<at::native::(anonymous namespace)::NonZeroOp<bool>, bool const*, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default>, int*, unsigned long long, at_cuda_detail::cub::GridEvenShare<unsigned long long>, cuda::std::__4::plus<void>, cuda::std::__4::__identity)": 4,
|
| 29 |
+
"void at_cuda_detail::cub::detail::reduce::DeviceReduceSingleTileKernel<at_cuda_detail::cub::detail::reduce::policy_hub<int, unsigned long long, cuda::std::__4::plus<void> >::Policy1000, int*, int*, int, cuda::std::__4::plus<void>, int, int, cuda::std::__4::__identity>(int*, int*, int, cuda::std::__4::plus<void>, int, cuda::std::__4::__identity)": 4,
|
| 30 |
+
"void at_cuda_detail::cub::detail::scan::DeviceCompactInitKernel<at_cuda_detail::cub::ScanTileState<int, true>, int*>(at_cuda_detail::cub::ScanTileState<int, true>, int, int*)": 4,
|
| 31 |
+
"void at_cuda_detail::cub::detail::select::DeviceSelectSweepKernel<at_cuda_detail::cub::detail::select::policy_hub<long, bool, int, false, (at_cuda_detail::cub::SelectImpl)0>::Policy1000, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::counting_iterator<long, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default>, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::transform_iterator<at::native::(anonymous namespace)::NonZeroOp<bool>, bool const*, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default>, long*, int*, at_cuda_detail::cub::ScanTileState<int, true>, at_cuda_detail::cub::NullType, at_cuda_detail::cub::NullType, int, at_cuda_detail::cub::detail::select::streaming_context_t<long, true>, (at_cuda_detail::cub::SelectImpl)0>(thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::counting_iterator<long, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default>, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::transform_iterator<at::native::(anonymous namespace)::NonZeroOp<bool>, bool const*, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default>, long*, int*, at_cuda_detail::cub::ScanTileState<int, true>, at_cuda_detail::cub::NullType, at_cuda_detail::cub::NullType, int, int, at_cuda_detail::cub::detail::select::streaming_context_t<long, true>, at_cuda_detail::cub::detail::vsmem_t)": 4,
|
| 32 |
+
"void at::native::index_elementwise_kernel<128, 4, at::native::gpu_index_kernel<at::native::index_put_kernel_impl<at::native::OpaqueType<2> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1}>(at::TensorIteratorBase&, c10::ArrayRef<long>, c10::ArrayRef<long>, at::native::index_put_kernel_impl<at::native::OpaqueType<2> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1} const&, bool)::{lambda(int)#1}>(long, at::native::gpu_index_kernel<at::native::index_put_kernel_impl<at::native::OpaqueType<2> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1}>(at::TensorIteratorBase&, c10::ArrayRef<long>, c10::ArrayRef<long>, at::native::index_put_kernel_impl<at::native::OpaqueType<2> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1} const&, bool)::{lambda(int)#1})": 1,
|
| 33 |
+
"Memcpy DtoH (Device -> Pageable)": 1,
|
| 34 |
+
"Memcpy HtoD (Pageable -> Device)": 5,
|
| 35 |
+
"Memcpy DtoD (Device -> Device)": 3,
|
| 36 |
+
"void at::native::index_elementwise_kernel<128, 4, at::native::gpu_index_kernel<at::native::index_put_kernel_impl<at::native::OpaqueType<8> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1}>(at::TensorIteratorBase&, c10::ArrayRef<long>, c10::ArrayRef<long>, at::native::index_put_kernel_impl<at::native::OpaqueType<8> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1} const&, bool)::{lambda(int)#1}>(long, at::native::gpu_index_kernel<at::native::index_put_kernel_impl<at::native::OpaqueType<8> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1}>(at::TensorIteratorBase&, c10::ArrayRef<long>, c10::ArrayRef<long>, at::native::index_put_kernel_impl<at::native::OpaqueType<8> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1} const&, bool)::{lambda(int)#1})": 3,
|
| 37 |
+
"void at::native::(anonymous namespace)::CatArrayBatchedCopy_vectorized<at::native::(anonymous namespace)::OpaqueType<8u>, unsigned int, 2, 128, 1, 16, 2>(char*, at::native::(anonymous namespace)::CatArrInputTensorMetadata<at::native::(anonymous namespace)::OpaqueType<8u>, unsigned int, 128, 1>, at::native::(anonymous namespace)::TensorSizeStride<unsigned int, 4u>, int, unsigned int)": 1,
|
| 38 |
+
"void (anonymous namespace)::elementwise_kernel_with_index<int, at::native::arange_cuda_out(c10::Scalar const&, c10::Scalar const&, c10::Scalar const&, at::Tensor&)::{lambda()#1}::operator()() const::{lambda()#4}::operator()() const::{lambda(long)#1}>(int, at::native::arange_cuda_out(c10::Scalar const&, c10::Scalar const&, c10::Scalar const&, at::Tensor&)::{lambda()#1}::operator()() const::{lambda()#4}::operator()() const::{lambda(long)#1}, function_traits<at::native::arange_cuda_out(c10::Scalar const&, c10::Scalar const&, c10::Scalar const&, at::Tensor&)::{lambda()#1}::operator()() const::{lambda()#4}::operator()() const::{lambda(long)#1}>::result_type*)": 1,
|
| 39 |
+
"void at::native::_scatter_gather_elementwise_kernel<128, 8, at::native::_cuda_scatter_gather_internal_kernel<false, at::native::OpaqueType<8>, long>::operator()<at::native::TensorAssign>(at::TensorIterator&, long, long, long, at::native::TensorAssign const&)::{lambda(int)#1}>(int, at::native::_cuda_scatter_gather_internal_kernel<false, at::native::OpaqueType<8>, long>::operator()<at::native::TensorAssign>(at::TensorIterator&, long, long, long, at::native::TensorAssign const&)::{lambda(int)#1})": 1,
|
| 40 |
+
"void at::native::vectorized_elementwise_kernel<4, at::native::FillFunctor<bool>, std::array<char*, 1ul>, false>(int, at::native::FillFunctor<bool>, std::array<char*, 1ul>)": 1,
|
| 41 |
+
"void at::native::index_elementwise_kernel<128, 4, at::native::gpu_index_kernel<at::native::index_put_kernel_impl<at::native::OpaqueType<1> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1}>(at::TensorIteratorBase&, c10::ArrayRef<long>, c10::ArrayRef<long>, at::native::index_put_kernel_impl<at::native::OpaqueType<1> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1} const&, bool)::{lambda(int)#1}>(long, at::native::gpu_index_kernel<at::native::index_put_kernel_impl<at::native::OpaqueType<1> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1}>(at::TensorIteratorBase&, c10::ArrayRef<long>, c10::ArrayRef<long>, at::native::index_put_kernel_impl<at::native::OpaqueType<1> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1} const&, bool)::{lambda(int)#1})": 1,
|
| 42 |
+
"void at::native::(anonymous namespace)::CatArrayBatchedCopy_alignedK_contig<at::native::(anonymous namespace)::OpaqueType<2u>, unsigned int, 1, 128, 1, 8>(at::native::(anonymous namespace)::OpaqueType<2u>*, at::native::(anonymous namespace)::CatArrInputTensorMetadata<at::native::(anonymous namespace)::OpaqueType<2u>, unsigned int, 128, 1>, at::native::(anonymous namespace)::TensorSizeStride<unsigned int, 4u>, int, unsigned int)": 1,
|
| 43 |
+
"void at::native::vectorized_elementwise_kernel<4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float> >, std::array<char*, 2ul>, false>(int, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float> >, std::array<char*, 2ul>)": 1,
|
| 44 |
+
"void at::native::vectorized_elementwise_kernel<4, at::native::cos_kernel_cuda(at::TensorIteratorBase&)::{lambda()#2}::operator()() const::{lambda()#2}::operator()() const::{lambda(float)#1}, std::array<char*, 2ul>, false>(int, at::native::cos_kernel_cuda(at::TensorIteratorBase&)::{lambda()#2}::operator()() const::{lambda()#2}::operator()() const::{lambda(float)#1}, std::array<char*, 2ul>)": 1,
|
| 45 |
+
"void at::native::vectorized_elementwise_kernel<4, at::native::sin_kernel_cuda(at::TensorIteratorBase&)::{lambda()#2}::operator()() const::{lambda()#2}::operator()() const::{lambda(float)#1}, std::array<char*, 2ul>, false>(int, at::native::sin_kernel_cuda(at::TensorIteratorBase&)::{lambda()#2}::operator()() const::{lambda()#2}::operator()() const::{lambda(float)#1}, std::array<char*, 2ul>)": 1,
|
| 46 |
+
"void at::native::(anonymous namespace)::CatArrayBatchedCopy_vectorized<at::native::(anonymous namespace)::OpaqueType<4u>, unsigned int, 2, 128, 1, 16, 4>(char*, at::native::(anonymous namespace)::CatArrInputTensorMetadata<at::native::(anonymous namespace)::OpaqueType<4u>, unsigned int, 128, 1>, at::native::(anonymous namespace)::TensorSizeStride<unsigned int, 4u>, int, unsigned int)": 1,
|
| 47 |
+
"void cutlass::Kernel2<cutlass_80_wmma_tensorop_bf16_s161616gemm_bf16_32x32_128x2_tn_align8>(cutlass_80_wmma_tensorop_bf16_s161616gemm_bf16_32x32_128x2_tn_align8::Params)": 1,
|
| 48 |
+
"void at::native::vectorized_elementwise_kernel<4, at::native::(anonymous namespace)::silu_kernel(at::TensorIteratorBase&)::{lambda()#1}::operator()() const::{lambda()#6}::operator()() const::{lambda(c10::BFloat16)#1}, std::array<char*, 2ul>, false>(int, at::native::(anonymous namespace)::silu_kernel(at::TensorIteratorBase&)::{lambda()#1}::operator()() const::{lambda()#6}::operator()() const::{lambda(c10::BFloat16)#1}, std::array<char*, 2ul>)": 3,
|
| 49 |
+
"Memset (Device)": 3,
|
| 50 |
+
"void cutlass::Kernel2<cutlass_80_wmma_tensorop_bf16_s161616gemm_bf16_32x32_128x1_tn_align8>(cutlass_80_wmma_tensorop_bf16_s161616gemm_bf16_32x32_128x1_tn_align8::Params)": 2,
|
| 51 |
+
"void cutlass::Kernel2<cutlass_80_wmma_tensorop_bf16_s161616gemm_bf16_32x32_64x1_tn_align8>(cutlass_80_wmma_tensorop_bf16_s161616gemm_bf16_32x32_64x1_tn_align8::Params)": 1,
|
| 52 |
+
"void at::native::vectorized_elementwise_kernel<4, at::native::bitwise_not_kernel_cuda(at::TensorIteratorBase&)::{lambda(bool)#1}, std::array<char*, 2ul>, false>(int, at::native::bitwise_not_kernel_cuda(at::TensorIteratorBase&)::{lambda(bool)#1}, std::array<char*, 2ul>)": 1,
|
| 53 |
+
"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda()#3}::operator()() const::{lambda()#4}::operator()() const::{lambda(long)#1}, std::array<char*, 2ul>, 4, TrivialOffsetCalculator<1, unsigned int>, TrivialOffsetCalculator<1, unsigned int>, at::native::memory::LoadWithCast<1>, at::native::memory::StoreWithCast<1> >(int, at::native::direct_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda()#3}::operator()() const::{lambda()#4}::operator()() const::{lambda(long)#1}, std::array<char*, 2ul>, TrivialOffsetCalculator<1, unsigned int>, TrivialOffsetCalculator<1, unsigned int>, at::native::memory::LoadWithCast<1>, at::native::memory::StoreWithCast<1>)": 1,
|
| 54 |
+
"void at::native::reduce_kernel<512, 1, at::native::ReduceOp<long, at::native::func_wrapper_t<long, at::native::sum_functor<long, long, long>::operator()(at::TensorIterator&)::{lambda(long, long)#1}>, unsigned int, long, 4, 4> >(at::native::ReduceOp<long, at::native::func_wrapper_t<long, at::native::sum_functor<long, long, long>::operator()(at::TensorIterator&)::{lambda(long, long)#1}>, unsigned int, long, 4, 4>)": 1,
|
| 55 |
+
"triton_red_fused_add_mul_native_layer_norm_slice_split_unsqueeze_0": 32,
|
| 56 |
+
"_quantize_rows": 224,
|
| 57 |
+
"_ZN7cutlass13device_kernelIN2at4cuda6detail34enable_3x_kernel_for_sm10_or_laterINS_4gemm6kernel13GemmUniversalIN4cute5tupleIJiiiEEENS5_10collective13CollectiveMmaINS5_31MainloopSm120TmaWarpSpecializedILi2ELi2ENS9_IJNS8_1CILi1EEESF_SF_EEENS5_40KernelTmaWarpSpecializedCooperativeSm120ILi2EEEEENS9_IJNSE_ILi128EEESK_SK_EEENS_12float_e4m3_tENS9_IJlSF_lEEESM_SN_NS8_8TiledMMAINS8_8MMA_AtomIJNS8_16SM120_16x8x32_TNISM_SM_fEEEEENS8_6LayoutINS9_IJNSE_ILi4EEENSE_ILi2EEESF_EEENS9_IJSF_SU_NSE_ILi0EEEEEEEENS9_IJSK_NSE_ILi32EEES10_EEEEENS8_13SM90_TMA_LOADENS8_14ComposedLayoutINS8_7SwizzleILi3ELi4ELi3EEENS8_18smem_ptr_flag_bitsILi8EEENST_INS9_IJNSE_ILi8EEESK_EEENS9_IJSK_SF_EEEEEEENS8_9Copy_AtomIJNS8_17SM75_U32x4_LDSM_NEhEEENS8_8identityES13_S1D_S1G_S1H_EENS_8epilogue10collective18CollectiveEpilogueINS1J_22Sm90TmaWarpSpecializedILi2ELi2ELi4ELb0ELb1EEEJSL_NS9_IJNSE_ILi64EEES10_EEEvSN_NS_10bfloat16_tESN_NS1J_6fusion15Sm90TreeVisitorINS1R_11Sm90ComputeINS1J_6thread8IdentityES1Q_fLNS_15FloatRoundStyleE2EvEEJNS1S_INS1T_INS_4plusEffLS1W_2EvEEJNS1R_16Sm90RowBroadcastILi0ESL_ffNS9_IJSX_SF_SX_EEELi4ELb1EEENS1S_INS1T_INS_10multipliesEffLS1W_2EvEEJS22_NS1S_IS24_JNS1R_16Sm90ColBroadcastILi0ESL_ffNS9_IJSF_SX_SX_EEELi4ELb1EEENS1R_12Sm90AccFetchEEEEEEEEEEEEES13_NS14_INS15_ILi2ELi4ELi3EEENS17_ILi16EEENST_INS9_IJS19_S10_EEENS9_IJS10_SF_EEEEEEENS8_17SM75_U32x2_LDSM_NENS8_14SM90_TMA_STOREES2I_NS8_17SM90_U32x2_STSM_NENS1E_IJS2L_NS_6half_tEEEEvEEEvvEEEEEEvNT_6ParamsE": 224,
|
| 58 |
+
"triton_per_fused__to_copy_mean_pow_view_1": 64,
|
| 59 |
+
"triton_poi_fused__to_copy_add_mean_mul_pow_rsqrt_select_sub_unsqueeze_view_2": 64,
|
| 60 |
+
"triton_poi_fused__scaled_dot_product_flash_attention__to_copy_cat_stack_transpose_view_3": 32,
|
| 61 |
+
"triton_poi_fused__scaled_dot_product_flash_attention__to_copy_cat_stack_transpose_view_4": 32,
|
| 62 |
+
"triton_poi_fused__scaled_dot_product_flash_attention__to_copy_cat_stack_transpose_view_5": 32,
|
| 63 |
+
"void pytorch_flash::flash_fwd_kernel<Flash_fwd_kernel_traits<128, 128, 64, 4, false, false, cutlass::bfloat16_t, Flash_kernel_traits<128, 128, 64, 4, cutlass::bfloat16_t> >, false, false, false, false, false, true, false, false>(pytorch_flash::Flash_fwd_params)": 32,
|
| 64 |
+
"triton_red_fused_add_mul_native_layer_norm_slice_split_tanh_unsqueeze_view_6": 32,
|
| 65 |
+
"triton_poi_fused_mul_silu_view_7": 32,
|
| 66 |
+
"triton_poi_fused_add_mul_slice_split_tanh_unsqueeze_view_8": 32,
|
| 67 |
+
"void at::native::elementwise_kernel<128, 4, at::native::gpu_kernel_impl_nocast<at::native::(anonymous namespace)::where_kernel_impl(at::TensorIterator&)::{lambda()#1}::operator()() const::{lambda()#2}::operator()() const::{lambda(bool, unsigned short, unsigned short)#1}>(at::TensorIteratorBase&, at::native::(anonymous namespace)::where_kernel_impl(at::TensorIterator&)::{lambda()#1}::operator()() const::{lambda()#2}::operator()() const::{lambda(bool, unsigned short, unsigned short)#1} const&)::{lambda(int)#1}>(int, at::native::gpu_kernel_impl_nocast<at::native::(anonymous namespace)::where_kernel_impl(at::TensorIterator&)::{lambda()#1}::operator()() const::{lambda()#2}::operator()() const::{lambda(bool, unsigned short, unsigned short)#1}>(at::TensorIteratorBase&, at::native::(anonymous namespace)::where_kernel_impl(at::TensorIterator&)::{lambda()#1}::operator()() const::{lambda()#2}::operator()() const::{lambda(bool, unsigned short, unsigned short)#1} const&)::{lambda(int)#1})": 1,
|
| 68 |
+
"void at::native::(anonymous namespace)::vectorized_layer_norm_kernel<c10::BFloat16, float, false>(int, float, c10::BFloat16 const*, c10::BFloat16 const*, c10::BFloat16 const*, float*, float*, c10::BFloat16*)": 1,
|
| 69 |
+
"void at::native::vectorized_elementwise_kernel<4, at::native::CUDAFunctorOnSelf_add<c10::BFloat16>, std::array<char*, 2ul>, false>(int, at::native::CUDAFunctorOnSelf_add<c10::BFloat16>, std::array<char*, 2ul>)": 1,
|
| 70 |
+
"void at::native::vectorized_elementwise_kernel<4, at::native::BinaryFunctor<c10::BFloat16, c10::BFloat16, c10::BFloat16, at::native::binary_internal::MulFunctor<float> >, std::array<char*, 3ul>, false>(int, at::native::BinaryFunctor<c10::BFloat16, c10::BFloat16, c10::BFloat16, at::native::binary_internal::MulFunctor<float> >, std::array<char*, 3ul>)": 1,
|
| 71 |
+
"void cutlass::Kernel2<cutlass_80_tensorop_bf16_s16816gemm_relu_bf16_64x128_64x3_tn_align8>(cutlass_80_tensorop_bf16_s16816gemm_relu_bf16_64x128_64x3_tn_align8::Params)": 1
|
| 72 |
+
},
|
| 73 |
+
"transformer_calls": 40,
|
| 74 |
+
"full_denoising_steps": 40
|
| 75 |
+
}
|
optimization/qa-review.json
ADDED
|
@@ -0,0 +1,16 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"visual_review_complete": true,
|
| 3 |
+
"pairs_reviewed": 18,
|
| 4 |
+
"observation": "No obvious general visual quality loss was observed in all 18 paired images, including portraits, wildlife, macro, food, architecture, paintings, glass, primary English and Chinese titles, transparency, crafts, coastline, flowers, human hands, and two edits. Outputs are not bit-identical: potter arm and clay positioning, snow leopard coat and background details, decorative typography, and some fine textures differ. The pottery sample has the largest LPIPS difference (0.10415) and remains visually plausible. This finite review cannot guarantee every prompt.",
|
| 5 |
+
"quality_priority": "User explicitly required quality not be hurt. Approximate residual caching and extra attentionquantization excluded fromfinalruntime. NativeBF16attention, full40steps, originalFP8weights/scales andFP32GEMMaccumulation kept. Compiledfusions emulateintermediateBF16castboundaries.",
|
| 6 |
+
"mean_lpips_vs_original_fp8": 0.014105255333965437,
|
| 7 |
+
"mean_ssim_vs_original_fp8": 0.9858412672397402,
|
| 8 |
+
"max_lpips_vs_original_fp8": 0.10415362566709518,
|
| 9 |
+
"kernel_verified": true,
|
| 10 |
+
"video_verified": true,
|
| 11 |
+
"timing_verified": true,
|
| 12 |
+
"kernel_observation": "Actual compiled denoising-step profile:224 native CUTLASS SM120 E4M3 weight GEMMs,32 native BF16 FlashAttention launches,40 transformer calls for40steps.",
|
| 13 |
+
"approved_for_publication": true,
|
| 14 |
+
"video_observation": "Verified30seconds,900frames,30fps,1024x1024,noaudio. Five newimages complete live by29.34seconds after oneinitialwarmup image. No timecompression. DiagnosticPNGcompression deferred until recordingends; maximum capturelag1.72ms.",
|
| 15 |
+
"video_sha256": "3271da65a4797dda30419d7857612fb05873497f2c50b18056c96fde92fb9457"
|
| 16 |
+
}
|
optimization/quality-vs-bf16.json
ADDED
|
@@ -0,0 +1,168 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"protocol": "16 disjoint text-to-image prompts plus2 disjoint edits. Matched seeds,40steps,same pipeline and BF16 encoder/VAE. RGB composited over white and resized512 for LPIPS/SSIM/PSNR. Latents compared at full resolution. This is a finite fidelity check, not a broad human preference benchmark.",
|
| 3 |
+
"pairs": [
|
| 4 |
+
{
|
| 5 |
+
"id": "00",
|
| 6 |
+
"lpips_alex_512": 0.00756214139983058,
|
| 7 |
+
"ssim_512": 0.9937157078162903,
|
| 8 |
+
"psnr_512": 40.75073049575512,
|
| 9 |
+
"latent_nrmse": 0.034632131457328796,
|
| 10 |
+
"latent_cosine": 0.9993438720703125,
|
| 11 |
+
"alpha_mae": 0.0002727321549957874
|
| 12 |
+
},
|
| 13 |
+
{
|
| 14 |
+
"id": "01",
|
| 15 |
+
"lpips_alex_512": 0.009154626168310642,
|
| 16 |
+
"ssim_512": 0.9886946931497306,
|
| 17 |
+
"psnr_512": 36.185498510043146,
|
| 18 |
+
"latent_nrmse": 0.03882736712694168,
|
| 19 |
+
"latent_cosine": 0.9992272257804871,
|
| 20 |
+
"alpha_mae": 0.00010991937973920037
|
| 21 |
+
},
|
| 22 |
+
{
|
| 23 |
+
"id": "02",
|
| 24 |
+
"lpips_alex_512": 0.03219389542937279,
|
| 25 |
+
"ssim_512": 0.9567132715264104,
|
| 26 |
+
"psnr_512": 31.192773668152384,
|
| 27 |
+
"latent_nrmse": 0.06602165848016739,
|
| 28 |
+
"latent_cosine": 0.9978012442588806,
|
| 29 |
+
"alpha_mae": 0.00029312582576976104
|
| 30 |
+
},
|
| 31 |
+
{
|
| 32 |
+
"id": "03",
|
| 33 |
+
"lpips_alex_512": 0.06048927456140518,
|
| 34 |
+
"ssim_512": 0.9374954223909032,
|
| 35 |
+
"psnr_512": 26.951171321484626,
|
| 36 |
+
"latent_nrmse": 0.12097524851560593,
|
| 37 |
+
"latent_cosine": 0.9926716089248657,
|
| 38 |
+
"alpha_mae": 0.000203095230401731
|
| 39 |
+
},
|
| 40 |
+
{
|
| 41 |
+
"id": "04",
|
| 42 |
+
"lpips_alex_512": 0.01704142801463604,
|
| 43 |
+
"ssim_512": 0.9795077559573834,
|
| 44 |
+
"psnr_512": 35.354291962916335,
|
| 45 |
+
"latent_nrmse": 0.05102594196796417,
|
| 46 |
+
"latent_cosine": 0.9986280202865601,
|
| 47 |
+
"alpha_mae": 4.380544026692708e-05
|
| 48 |
+
},
|
| 49 |
+
{
|
| 50 |
+
"id": "05",
|
| 51 |
+
"lpips_alex_512": 0.04030201584100723,
|
| 52 |
+
"ssim_512": 0.9757162792169272,
|
| 53 |
+
"psnr_512": 30.341329853960055,
|
| 54 |
+
"latent_nrmse": 0.06395237147808075,
|
| 55 |
+
"latent_cosine": 0.9979398250579834,
|
| 56 |
+
"alpha_mae": 0.0002221051384420956
|
| 57 |
+
},
|
| 58 |
+
{
|
| 59 |
+
"id": "06",
|
| 60 |
+
"lpips_alex_512": 0.006186304148286581,
|
| 61 |
+
"ssim_512": 0.9907699698891504,
|
| 62 |
+
"psnr_512": 34.73095417820288,
|
| 63 |
+
"latent_nrmse": 0.03467632457613945,
|
| 64 |
+
"latent_cosine": 0.9993797540664673,
|
| 65 |
+
"alpha_mae": 0.0004366332409428615
|
| 66 |
+
},
|
| 67 |
+
{
|
| 68 |
+
"id": "07",
|
| 69 |
+
"lpips_alex_512": 0.0854998379945755,
|
| 70 |
+
"ssim_512": 0.9284355594200152,
|
| 71 |
+
"psnr_512": 28.733093561839304,
|
| 72 |
+
"latent_nrmse": 0.15828152000904083,
|
| 73 |
+
"latent_cosine": 0.9874526262283325,
|
| 74 |
+
"alpha_mae": 5.0724253934972426e-05
|
| 75 |
+
},
|
| 76 |
+
{
|
| 77 |
+
"id": "08",
|
| 78 |
+
"lpips_alex_512": 0.006477963179349899,
|
| 79 |
+
"ssim_512": 0.9891401459316924,
|
| 80 |
+
"psnr_512": 31.9284938234566,
|
| 81 |
+
"latent_nrmse": 0.03743818402290344,
|
| 82 |
+
"latent_cosine": 0.9992266297340393,
|
| 83 |
+
"alpha_mae": 6.847755581724878e-06
|
| 84 |
+
},
|
| 85 |
+
{
|
| 86 |
+
"id": "09",
|
| 87 |
+
"lpips_alex_512": 0.0163662638515234,
|
| 88 |
+
"ssim_512": 0.9870575264050417,
|
| 89 |
+
"psnr_512": 38.15120347682023,
|
| 90 |
+
"latent_nrmse": 0.04842615872621536,
|
| 91 |
+
"latent_cosine": 0.9988112449645996,
|
| 92 |
+
"alpha_mae": 4.506578632429534e-06
|
| 93 |
+
},
|
| 94 |
+
{
|
| 95 |
+
"id": "10",
|
| 96 |
+
"lpips_alex_512": 0.0033354151528328657,
|
| 97 |
+
"ssim_512": 0.9935745573136584,
|
| 98 |
+
"psnr_512": 36.583516491471485,
|
| 99 |
+
"latent_nrmse": 0.024394527077674866,
|
| 100 |
+
"latent_cosine": 0.9996834993362427,
|
| 101 |
+
"alpha_mae": 0.002104351567287071
|
| 102 |
+
},
|
| 103 |
+
{
|
| 104 |
+
"id": "11",
|
| 105 |
+
"lpips_alex_512": 0.0451497957110405,
|
| 106 |
+
"ssim_512": 0.9747825856715794,
|
| 107 |
+
"psnr_512": 28.746297060604743,
|
| 108 |
+
"latent_nrmse": 0.07993289083242416,
|
| 109 |
+
"latent_cosine": 0.9967884421348572,
|
| 110 |
+
"alpha_mae": 4.8880483589920346e-05
|
| 111 |
+
},
|
| 112 |
+
{
|
| 113 |
+
"id": "12",
|
| 114 |
+
"lpips_alex_512": 0.100985586643219,
|
| 115 |
+
"ssim_512": 0.9409372160895463,
|
| 116 |
+
"psnr_512": 21.850298325810687,
|
| 117 |
+
"latent_nrmse": 0.2208797037601471,
|
| 118 |
+
"latent_cosine": 0.9756127595901489,
|
| 119 |
+
"alpha_mae": 0.00015549707054889856
|
| 120 |
+
},
|
| 121 |
+
{
|
| 122 |
+
"id": "13",
|
| 123 |
+
"lpips_alex_512": 0.02744799479842186,
|
| 124 |
+
"ssim_512": 0.9734027147306175,
|
| 125 |
+
"psnr_512": 31.693602971981562,
|
| 126 |
+
"latent_nrmse": 0.06540966033935547,
|
| 127 |
+
"latent_cosine": 0.9978395104408264,
|
| 128 |
+
"alpha_mae": 1.0052849264705882e-05
|
| 129 |
+
},
|
| 130 |
+
{
|
| 131 |
+
"id": "14",
|
| 132 |
+
"lpips_alex_512": 0.09300243109464645,
|
| 133 |
+
"ssim_512": 0.9047834561435563,
|
| 134 |
+
"psnr_512": 24.483383727237495,
|
| 135 |
+
"latent_nrmse": 0.15774944424629211,
|
| 136 |
+
"latent_cosine": 0.9876120686531067,
|
| 137 |
+
"alpha_mae": 0.00021505168839996936
|
| 138 |
+
},
|
| 139 |
+
{
|
| 140 |
+
"id": "15",
|
| 141 |
+
"lpips_alex_512": 0.03884485363960266,
|
| 142 |
+
"ssim_512": 0.9593329727305248,
|
| 143 |
+
"psnr_512": 26.405222735216963,
|
| 144 |
+
"latent_nrmse": 0.10906766355037689,
|
| 145 |
+
"latent_cosine": 0.9940686225891113,
|
| 146 |
+
"alpha_mae": 0.00011861464556525735
|
| 147 |
+
},
|
| 148 |
+
{
|
| 149 |
+
"id": "edit-0",
|
| 150 |
+
"lpips_alex_512": 0.04596696048974991,
|
| 151 |
+
"ssim_512": 0.9749776380201428,
|
| 152 |
+
"psnr_512": 31.627062214355384,
|
| 153 |
+
"alpha_mae": 0.0001376507329005821
|
| 154 |
+
},
|
| 155 |
+
{
|
| 156 |
+
"id": "edit-1",
|
| 157 |
+
"lpips_alex_512": 0.0018672043224796653,
|
| 158 |
+
"ssim_512": 0.9943739435943822,
|
| 159 |
+
"psnr_512": 43.051310379117744,
|
| 160 |
+
"alpha_mae": 4.151288200827206e-05
|
| 161 |
+
}
|
| 162 |
+
],
|
| 163 |
+
"mean_lpips": 0.0354374440244606,
|
| 164 |
+
"max_lpips": 0.100985586643219,
|
| 165 |
+
"mean_ssim": 0.969078411999864,
|
| 166 |
+
"min_ssim": 0.9047834561435563,
|
| 167 |
+
"mean_latent_cosine": 0.9951304346323013
|
| 168 |
+
}
|
optimization/quality-vs-fp8.json
ADDED
|
@@ -0,0 +1,168 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"protocol": "16 disjoint text-to-image prompts plus2 disjoint edits. Matched seeds,40steps,same pipeline and BF16 encoder/VAE. RGB composited over white and resized512 for LPIPS/SSIM/PSNR. Latents compared at full resolution. This is a finite fidelity check, not a broad human preference benchmark.",
|
| 3 |
+
"pairs": [
|
| 4 |
+
{
|
| 5 |
+
"id": "00",
|
| 6 |
+
"lpips_alex_512": 0.0006540275062434375,
|
| 7 |
+
"ssim_512": 0.9983653564145475,
|
| 8 |
+
"psnr_512": 50.45740191583909,
|
| 9 |
+
"latent_nrmse": 0.008030623197555542,
|
| 10 |
+
"latent_cosine": 0.999891996383667,
|
| 11 |
+
"alpha_mae": 0.00022723815020392923
|
| 12 |
+
},
|
| 13 |
+
{
|
| 14 |
+
"id": "01",
|
| 15 |
+
"lpips_alex_512": 0.0032667126506567,
|
| 16 |
+
"ssim_512": 0.9941584529567485,
|
| 17 |
+
"psnr_512": 40.6246243151626,
|
| 18 |
+
"latent_nrmse": 0.022569432854652405,
|
| 19 |
+
"latent_cosine": 0.9997267723083496,
|
| 20 |
+
"alpha_mae": 8.504156972847733e-05
|
| 21 |
+
},
|
| 22 |
+
{
|
| 23 |
+
"id": "02",
|
| 24 |
+
"lpips_alex_512": 0.01659783348441124,
|
| 25 |
+
"ssim_512": 0.9809412274934068,
|
| 26 |
+
"psnr_512": 35.86132382974127,
|
| 27 |
+
"latent_nrmse": 0.04317404329776764,
|
| 28 |
+
"latent_cosine": 0.9990493059158325,
|
| 29 |
+
"alpha_mae": 0.00024381151386335784
|
| 30 |
+
},
|
| 31 |
+
{
|
| 32 |
+
"id": "03",
|
| 33 |
+
"lpips_alex_512": 0.005000461358577013,
|
| 34 |
+
"ssim_512": 0.9900833520273945,
|
| 35 |
+
"psnr_512": 38.8444442873744,
|
| 36 |
+
"latent_nrmse": 0.024173730984330177,
|
| 37 |
+
"latent_cosine": 0.9996890425682068,
|
| 38 |
+
"alpha_mae": 0.00014505947337431065
|
| 39 |
+
},
|
| 40 |
+
{
|
| 41 |
+
"id": "04",
|
| 42 |
+
"lpips_alex_512": 0.000704753038007766,
|
| 43 |
+
"ssim_512": 0.9979097545027972,
|
| 44 |
+
"psnr_512": 47.51767239675964,
|
| 45 |
+
"latent_nrmse": 0.011123890988528728,
|
| 46 |
+
"latent_cosine": 0.9998582601547241,
|
| 47 |
+
"alpha_mae": 3.1303891948625156e-05
|
| 48 |
+
},
|
| 49 |
+
{
|
| 50 |
+
"id": "05",
|
| 51 |
+
"lpips_alex_512": 0.008891566656529903,
|
| 52 |
+
"ssim_512": 0.9927372745460277,
|
| 53 |
+
"psnr_512": 35.948886615374875,
|
| 54 |
+
"latent_nrmse": 0.029095789417624474,
|
| 55 |
+
"latent_cosine": 0.9995598793029785,
|
| 56 |
+
"alpha_mae": 0.00015900555778952205
|
| 57 |
+
},
|
| 58 |
+
{
|
| 59 |
+
"id": "06",
|
| 60 |
+
"lpips_alex_512": 0.0013984774705022573,
|
| 61 |
+
"ssim_512": 0.9956754928433855,
|
| 62 |
+
"psnr_512": 45.45275428180727,
|
| 63 |
+
"latent_nrmse": 0.010114307515323162,
|
| 64 |
+
"latent_cosine": 0.9999299049377441,
|
| 65 |
+
"alpha_mae": 0.00036374260397518385
|
| 66 |
+
},
|
| 67 |
+
{
|
| 68 |
+
"id": "07",
|
| 69 |
+
"lpips_alex_512": 0.016953393816947937,
|
| 70 |
+
"ssim_512": 0.9791173071202123,
|
| 71 |
+
"psnr_512": 36.43153122071263,
|
| 72 |
+
"latent_nrmse": 0.04522187262773514,
|
| 73 |
+
"latent_cosine": 0.9989587068557739,
|
| 74 |
+
"alpha_mae": 3.706614176432292e-05
|
| 75 |
+
},
|
| 76 |
+
{
|
| 77 |
+
"id": "08",
|
| 78 |
+
"lpips_alex_512": 0.014839287847280502,
|
| 79 |
+
"ssim_512": 0.9915602272992333,
|
| 80 |
+
"psnr_512": 32.45296744802643,
|
| 81 |
+
"latent_nrmse": 0.03516024351119995,
|
| 82 |
+
"latent_cosine": 0.9993060231208801,
|
| 83 |
+
"alpha_mae": 5.567775053136489e-06
|
| 84 |
+
},
|
| 85 |
+
{
|
| 86 |
+
"id": "09",
|
| 87 |
+
"lpips_alex_512": 0.02529650554060936,
|
| 88 |
+
"ssim_512": 0.9808640721341937,
|
| 89 |
+
"psnr_512": 34.14858031516565,
|
| 90 |
+
"latent_nrmse": 0.05903178080916405,
|
| 91 |
+
"latent_cosine": 0.9982385635375977,
|
| 92 |
+
"alpha_mae": 4.1288488051470586e-06
|
| 93 |
+
},
|
| 94 |
+
{
|
| 95 |
+
"id": "10",
|
| 96 |
+
"lpips_alex_512": 0.00157443608622998,
|
| 97 |
+
"ssim_512": 0.996625042145732,
|
| 98 |
+
"psnr_512": 40.29209168419813,
|
| 99 |
+
"latent_nrmse": 0.012434827163815498,
|
| 100 |
+
"latent_cosine": 0.9999027252197266,
|
| 101 |
+
"alpha_mae": 0.0015476002412683823
|
| 102 |
+
},
|
| 103 |
+
{
|
| 104 |
+
"id": "11",
|
| 105 |
+
"lpips_alex_512": 0.008122559636831284,
|
| 106 |
+
"ssim_512": 0.9937619445127668,
|
| 107 |
+
"psnr_512": 36.5806762030683,
|
| 108 |
+
"latent_nrmse": 0.04052620381116867,
|
| 109 |
+
"latent_cosine": 0.999162495136261,
|
| 110 |
+
"alpha_mae": 3.919040455537684e-05
|
| 111 |
+
},
|
| 112 |
+
{
|
| 113 |
+
"id": "12",
|
| 114 |
+
"lpips_alex_512": 0.003888127626851201,
|
| 115 |
+
"ssim_512": 0.996000746894162,
|
| 116 |
+
"psnr_512": 40.781900028166774,
|
| 117 |
+
"latent_nrmse": 0.019904406741261482,
|
| 118 |
+
"latent_cosine": 0.9997774362564087,
|
| 119 |
+
"alpha_mae": 9.341880560094409e-05
|
| 120 |
+
},
|
| 121 |
+
{
|
| 122 |
+
"id": "13",
|
| 123 |
+
"lpips_alex_512": 0.00787642877548933,
|
| 124 |
+
"ssim_512": 0.9938821301981401,
|
| 125 |
+
"psnr_512": 40.403632630082136,
|
| 126 |
+
"latent_nrmse": 0.030282530933618546,
|
| 127 |
+
"latent_cosine": 0.9995153546333313,
|
| 128 |
+
"alpha_mae": 6.808278867102396e-06
|
| 129 |
+
},
|
| 130 |
+
{
|
| 131 |
+
"id": "14",
|
| 132 |
+
"lpips_alex_512": 0.10415362566709518,
|
| 133 |
+
"ssim_512": 0.8997217393167819,
|
| 134 |
+
"psnr_512": 24.966557191764256,
|
| 135 |
+
"latent_nrmse": 0.1631270796060562,
|
| 136 |
+
"latent_cosine": 0.9867432713508606,
|
| 137 |
+
"alpha_mae": 0.0002106460870481005
|
| 138 |
+
},
|
| 139 |
+
{
|
| 140 |
+
"id": "15",
|
| 141 |
+
"lpips_alex_512": 0.032293133437633514,
|
| 142 |
+
"ssim_512": 0.9696098907103111,
|
| 143 |
+
"psnr_512": 27.556757057629092,
|
| 144 |
+
"latent_nrmse": 0.09265656769275665,
|
| 145 |
+
"latent_cosine": 0.9956912994384766,
|
| 146 |
+
"alpha_mae": 0.00012047337550742954
|
| 147 |
+
},
|
| 148 |
+
{
|
| 149 |
+
"id": "edit-0",
|
| 150 |
+
"lpips_alex_512": 0.001904923701658845,
|
| 151 |
+
"ssim_512": 0.9974066241437832,
|
| 152 |
+
"psnr_512": 43.80201499062957,
|
| 153 |
+
"alpha_mae": 9.189306520948223e-05
|
| 154 |
+
},
|
| 155 |
+
{
|
| 156 |
+
"id": "edit-1",
|
| 157 |
+
"lpips_alex_512": 0.0004783417098224163,
|
| 158 |
+
"ssim_512": 0.9967221750556993,
|
| 159 |
+
"psnr_512": 51.29105397500459,
|
| 160 |
+
"alpha_mae": 3.621718462775735e-05
|
| 161 |
+
}
|
| 162 |
+
],
|
| 163 |
+
"mean_lpips": 0.014105255333965437,
|
| 164 |
+
"max_lpips": 0.10415362566709518,
|
| 165 |
+
"mean_ssim": 0.9858412672397402,
|
| 166 |
+
"min_ssim": 0.8997217393167819,
|
| 167 |
+
"mean_latent_cosine": 0.9984375648200512
|
| 168 |
+
}
|
optimization/samples/00.png
ADDED
|
Git LFS Details
|
optimization/samples/01.png
ADDED
|
Git LFS Details
|
optimization/samples/02.png
ADDED
|
Git LFS Details
|
optimization/samples/03.png
ADDED
|
Git LFS Details
|
optimization/samples/04.png
ADDED
|
Git LFS Details
|
optimization/samples/05.png
ADDED
|
Git LFS Details
|
optimization/samples/06.png
ADDED
|
Git LFS Details
|
optimization/samples/07.png
ADDED
|
Git LFS Details
|
optimization/samples/08.png
ADDED
|
Git LFS Details
|
optimization/samples/09.png
ADDED
|
Git LFS Details
|
optimization/samples/10.png
ADDED
|
Git LFS Details
|
optimization/samples/11.png
ADDED
|
Git LFS Details
|
optimization/samples/12.png
ADDED
|
Git LFS Details
|
optimization/samples/13.png
ADDED
|
Git LFS Details
|
optimization/samples/14.png
ADDED
|
Git LFS Details
|
optimization/samples/15.png
ADDED
|
Git LFS Details
|
optimization/samples/edit-0.png
ADDED
|
Git LFS Details
|