{
 "generated_at": "2026-09-10T16:44:26+0800",
 "tag": "704x1280",
 "steady_kv": 63360,
 "peaks": {
  "bf16_dense_tflops": 989.5,
  "tf32_dense_tflops": 494.5,
  "fp32_tflops": 67.0
 },
 "dit": {
  "flops_per_chunk_all_ranks": 4493880223334400,
  "by_family": {
   "linear": 1731176064614400,
   "self_attention": 2740558233600000,
   "cross_attention": 22145925120000
  },
  "calls_by_family": {
   "linear": 8160,
   "self_attention": 800,
   "cross_attention": 800
  },
  "per_rank": {
   "0": 1123470055833600,
   "1": 1123470055833600,
   "2": 1123470055833600,
   "3": 1123470055833600
  },
  "elementwise_output_elements": 1481695568280,
  "forwards_per_chunk": 5,
  "wall_ms": 2138.2268679999997,
  "mfu_bf16_dense": 0.5309968204698888,
  "gpu_busy_ms_median": {
   "4": 2039.8515710000001,
   "5": 2036.0224715,
   "6": 2038.156337,
   "7": 2036.48578
  },
  "mfu_over_gpu_busy": 0.5572121549764231
 },
 "vae": {
  "flops_per_chunk_all_ranks": 91467774689280,
  "by_family": {
   "conv": 91467774689280
  },
  "calls_by_family": {
   "conv": 460
  },
  "per_rank": {
   "0": 22866943672320,
   "1": 22866943672320,
   "2": 22866943672320,
   "3": 22866943672320
  },
  "decode_ms": 290.0286515,
  "mfu_tf32_dense": 0.15944133524560092,
  "mfu_fp32_cuda_cores": 1.1767722429693976,
  "note": "decoder conv runs as TF32 implicit GEMM (xmma f32f32_tf32f32 kernels); TF32 dense peak is the matching denominator; FP32 CUDA-core figure shown for reference"
 },
 "end_to_end": {
  "delivery_interval_ms": 2483.617491,
  "fps": 4.831661897810334,
  "total_nominal_flops": 4585347998023680,
  "mfu_bf16_dense_reference": 0.4664572008222147,
  "tflops_per_s_per_gpu": 461.55940021358145
 },
 "scope": "Nominal dense FLOPs from eager dispatch shapes of the same configuration (frame 15, KV at steady length); times from the same-run ledger critical path (diagnostic trace, not clean timing). Hardware-executed FLOPs, tensor-core activity and per-kernel MFU are different metrics (see bottleneck cards)."
}
