{ "load_seconds": 2.297096138005145, "torch": "2.14.0+cu130", "gpu": "NVIDIA RTX PRO 6000 Blackwell Workstation Edition", "nvfp4_linears": 224, "steps": 40, "cfg": 1, "bf16_rank": 128, "attention_dtype": "bfloat16", "approximate_cache": false, "timing": { "1024": { "seconds": [ 4.547408219019417, 4.571850906999316, 4.592371508013457, 4.610020697989967, 4.624509044981096 ], "mean": 4.58923207540065, "warmup_seconds": 8.535866206977516, "peak_gb": 30.255306752 }, "2048": { "seconds": [ 32.58121982298326, 32.67262674000813, 32.71958359400742 ], "mean": 32.657810052332934, "warmup_seconds": 35.41050831298344, "peak_gb": 51.385951232 } }, "protocol": "CUDA synchronized; batch1; full40steps; includes encoder, denoising and VAE; excludes model load, resolution warmup and file writes. Prefix KV cache enabled. All large projections use either native NVFP4 with BF16 rank128 correction or explicitly listed calibrated FP8 safety layers. Compiled mode emulates intermediate precision casts." }