{ "variant": "orbitquant", "hardware": "NVIDIA A40", "source_revision": "98e0fe118d17c9e3547fbb2e25acdbae2cadf7c7", "quantized_revision": "49e89c37f51cb394b4ee199196d9df8bcd042475", "candidate_root": "/workspace/krea2-final-transformer-imgin", "settings": { "width": 2048, "height": 2048, "num_inference_steps": 8, "guidance_scale": 0.0, "seed_base": 61000, "dtype": "bfloat16", "attention_backend": "strict Flash SDPA", "prompt_compaction": "lossless valid-token compaction", "vae_tiling": [ 1024, 896 ], "component_policy": "Qwen encode then release; DiT and tiled VAE render" }, "load_seconds": 8.288543954491615, "load_gpu_peak_mb": 12827, "load_torch_peak_mb": 12432.95947265625, "qwen_inventory": { "class_name": "Qwen3VLModel", "orbitquant_linear_count": 356, "source_precision_linear_count": 0, "executed_orbitquant_linear_count": 252, "effective_runtime_modes": [ "native_packed_matmul" ], "activation_kernel_backends": [ "native_cuda_int8_surrogate", "triton_cuda_int8_surrogate" ], "last_activation_cache_hits": 108, "dequantized_cache_bytes": 0, "int8_weight_cache_bytes": 0 }, "dit_inventory": { "class_name": "Krea2Transformer2DModel", "orbitquant_linear_count": 232, "source_precision_linear_count": 32, "executed_orbitquant_linear_count": 232, "effective_runtime_modes": [ "native_packed_matmul" ], "activation_kernel_backends": [ "triton_cuda", "triton_cuda_int8_surrogate", "triton_cuda_packed_w4" ], "last_activation_cache_hits": 128, "dequantized_cache_bytes": 0, "int8_weight_cache_bytes": 0 }, "generated_samples": 1, "first_generation_seconds": 68.87463613599539, "hot_generation_mean_seconds": 68.87463613599539, "hot_generation_median_seconds": 68.87463613599539, "generation_mean_seconds": 68.87463613599539, "generation_median_seconds": 68.87463613599539, "gpu_peak_mb": 18657, "torch_peak_mb": 16792.41357421875 }