{
 "generated_at": "2026-09-10T16:44:25+0800",
 "tag": "480x832",
 "steady_kv": 28080,
 "peaks": {
  "bf16_dense_tflops": 989.5,
  "tf32_dense_tflops": 494.5,
  "fp32_tflops": 67.0
 },
 "dit": {
  "flops_per_chunk_all_ranks": 1315326158438400,
  "by_family": {
   "linear": 767238104678400,
   "self_attention": 538273382400000,
   "cross_attention": 9814671360000
  },
  "calls_by_family": {
   "linear": 8160,
   "self_attention": 800,
   "cross_attention": 800
  },
  "per_rank": {
   "0": 328831539609600,
   "1": 328831539609600,
   "2": 328831539609600,
   "3": 328831539609600
  },
  "elementwise_output_elements": 657267636120,
  "forwards_per_chunk": 5,
  "wall_ms": 677.5455489999999,
  "mfu_bf16_dense": 0.49047759172237615,
  "gpu_busy_ms_median": {
   "0": 602.77121,
   "1": 602.3830995000001,
   "2": 601.7894845,
   "3": 602.8283905000001
  },
  "mfu_over_gpu_busy": 0.551622118129292
 },
 "vae": {
  "flops_per_chunk_all_ranks": 40536854691840,
  "by_family": {
   "conv": 40536854691840
  },
  "calls_by_family": {
   "conv": 460
  },
  "per_rank": {
   "0": 10134213672960,
   "1": 10134213672960,
   "2": 10134213672960,
   "3": 10134213672960
  },
  "decode_ms": 141.5076885,
  "mfu_tf32_dense": 0.1448250623057939,
  "mfu_fp32_cuda_cores": 1.0688954225405234,
  "note": "decoder conv runs as TF32 implicit GEMM (xmma f32f32_tf32f32 kernels); TF32 dense peak is the matching denominator; FP32 CUDA-core figure shown for reference"
 },
 "end_to_end": {
  "delivery_interval_ms": 851.5501029999999,
  "fps": 14.091948269073255,
  "total_nominal_flops": 1355863013130240,
  "mfu_bf16_dense_reference": 0.4022812750753986,
  "tflops_per_s_per_gpu": 398.05732168710693
 },
 "scope": "Nominal dense FLOPs from eager dispatch shapes of the same configuration (frame 15, KV at steady length); times from the same-run ledger critical path (diagnostic trace, not clean timing). Hardware-executed FLOPs, tensor-core activity and per-kernel MFU are different metrics (see bottleneck cards)."
}
