{
 "generated_at": "2026-09-10T08:16:40.812108+00:00",
 "rules": "compute_bound: sm or tensor pipe >= 60%; dram_bandwidth_bound: dram >= 60%; l2_bandwidth_bound: lts >= 60% & dram < 60%; partial_wave_tail: sm_active/elapsed < 0.85 & waves <= 1.5; latency_or_occupancy_limited: all < 40% & (warps_active < 25% or waves < 1); else unresolved",
 "cards": [
  {
   "group": "dit:aten.linear.default:{'M': 4680, 'N': 5120, 'K': 144}",
   "resolution": "480x832",
   "stage": "dit",
   "op": "aten.linear.default",
   "shape_parameters": {
    "M": 4680,
    "N": 5120,
    "K": 144
   },
   "input_sha256": "824a77bcaead8af0dacf908827027966a81c2d964d919eb4200bac0771b4a369",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_capability_batch_dit_rank0_cd8cbc742f5c3f37_v1",
   "scheduler_note": null,
   "nominal_mac_flops": 6900940800,
   "summed_kernel_us": 25.312,
   "kernels": [
    {
     "kernel": "nvjet_sm90_tst_256x144_64x4_2x1_v_bz_coopA_bias_TNT",
     "metrics": {
      "dur_us": 25.312,
      "sm_pct": 38.789238,
      "tensor_pct": 38.789238,
      "dram_pct": 20.414169,
      "l2_pct": 56.807163,
      "warps_active_pct": 14.556988,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 221.38,
      "smem_static_b": 0.0,
      "l2_hit_pct": 96.985875,
      "sm_cycles_active": 38827.507576,
      "cycles_elapsed": 44786.0,
      "dram_read_mb": 2.86592,
      "dram_write_mb": 21.970688,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 38.789238,
      "tensor_pct": 38.789238,
      "dram_pct": 20.414169,
      "l2_pct": 56.807163,
      "warps_active_pct": 14.556988,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.8669563608270442
     }
    }
   ],
   "dominant_kernel": "nvjet_sm90_tst_256x144_64x4_2x1_v_bz_coopA_bias_TNT",
   "primary": "unresolved_no_dominant_limiter",
   "primary_rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
   "alternative_explanation": "Mixed limiter or measurement window artifact",
   "counter_experiment": "Collect stall reasons (smsp__warp_issue_stalled_*) and l1tex/lts bytes",
   "status": "unresolved"
  },
  {
   "group": "dit:aten.linear.default:{'M': 4680, 'N': 5120, 'K': 1536}",
   "resolution": "480x832",
   "stage": "dit",
   "op": "aten.linear.default",
   "shape_parameters": {
    "M": 4680,
    "N": 5120,
    "K": 1536
   },
   "input_sha256": "787b43c2c5ab957d3bd3c45047d646e6e407e506ea71b3249408d5d20bf00b0c",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_capability_batch_dit_rank0_9296b219534b05e0_v1",
   "scheduler_note": null,
   "nominal_mac_flops": 73610035200,
   "summed_kernel_us": 95.52,
   "kernels": [
    {
     "kernel": "nvjet_sm90_tst_256x144_64x4_2x1_v_bz_coopA_bias_TNT",
     "metrics": {
      "dur_us": 95.52,
      "sm_pct": 83.74599,
      "tensor_pct": 83.74599,
      "dram_pct": 17.378666,
      "l2_pct": 54.057958,
      "warps_active_pct": 14.734277,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 221.38,
      "smem_static_b": 0.0,
      "l2_hit_pct": 86.714984,
      "sm_cycles_active": 158954.181818,
      "cycles_elapsed": 167007.0,
      "dram_read_mb": 46.2784,
      "dram_write_mb": 33.586688,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 83.74599,
      "tensor_pct": 83.74599,
      "dram_pct": 17.378666,
      "l2_pct": 54.057958,
      "warps_active_pct": 14.734277,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.9517815529768214
     }
    }
   ],
   "dominant_kernel": "nvjet_sm90_tst_256x144_64x4_2x1_v_bz_coopA_bias_TNT",
   "primary": "compute_bound",
   "primary_rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
   "alternative_explanation": "Padding/redundant work inside nominal FLOPs; instruction mix vs tensor pipe",
   "counter_experiment": "Same dtype better tile/kernel; compare effective vs executed FLOPs",
   "status": "classified"
  },
  {
   "group": "dit:aten.linear.default:{'M': 4680, 'N': 5120, 'K': 5120}",
   "resolution": "480x832",
   "stage": "dit",
   "op": "aten.linear.default",
   "shape_parameters": {
    "M": 4680,
    "N": 5120,
    "K": 5120
   },
   "input_sha256": "4a5b492ee606c1a2c9fe4844a0c3125609fa599f4c027f56b4329c89d2879b7e",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_capability_dit_v1",
   "scheduler_note": null,
   "nominal_mac_flops": 245366784000,
   "summed_kernel_us": 292.4479999999999,
   "kernels": [
    {
     "kernel": "nvjet_sm90_tst_256x144_64x4_2x1_v_bz_coopA_bias_TNT",
     "metrics": {
      "dur_us": 292.448,
      "sm_pct": 94.027733,
      "tensor_pct": 94.027733,
      "dram_pct": 25.371858,
      "l2_pct": 52.627935,
      "warps_active_pct": 14.649706,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 221.38,
      "smem_static_b": 0.0,
      "l2_hit_pct": 74.505757,
      "sm_cycles_active": 482587.719697,
      "cycles_elapsed": 493471.0,
      "dram_read_mb": 314.121984,
      "dram_write_mb": 42.865664,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 94.027733,
      "tensor_pct": 94.027733,
      "dram_pct": 25.371858,
      "l2_pct": 52.627935,
      "warps_active_pct": 14.649706,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.9779454510943906
     }
    }
   ],
   "dominant_kernel": "nvjet_sm90_tst_256x144_64x4_2x1_v_bz_coopA_bias_TNT",
   "primary": "compute_bound",
   "primary_rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
   "alternative_explanation": "Padding/redundant work inside nominal FLOPs; instruction mix vs tensor pipe",
   "counter_experiment": "Same dtype better tile/kernel; compare effective vs executed FLOPs",
   "status": "classified"
  },
  {
   "group": "dit:aten.linear.default:{'M': 3, 'N': 5120, 'K': 256}",
   "resolution": "480x832",
   "stage": "dit",
   "op": "aten.linear.default",
   "shape_parameters": {
    "M": 3,
    "N": 5120,
    "K": 256
   },
   "input_sha256": "ea1184eaad12555ac80bf5ee33de7aa2251926b289572cadadb4043884848587",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_capability_batch_dit_rank0_10a11d3772070763_v1",
   "scheduler_note": null,
   "nominal_mac_flops": 7864320,
   "summed_kernel_us": 4.672,
   "kernels": [
    {
     "kernel": "nvjet_sm90_tst_64x8_64x16_4x1_v_bz_bias_TNT",
     "metrics": {
      "dur_us": 4.672,
      "sm_pct": 5.803488,
      "tensor_pct": 0.449718,
      "dram_pct": 11.795973,
      "l2_pct": 15.37494,
      "warps_active_pct": 14.14034,
      "max_warps_pct": 18.75,
      "waves": 0.61,
      "grid": 80.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 164.308,
      "smem_static_b": 0.0,
      "l2_hit_pct": 13.366767,
      "sm_cycles_active": 3245.166667,
      "cycles_elapsed": 8670.0,
      "dram_read_mb": 2.649344,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 5.803488,
      "tensor_pct": 0.449718,
      "dram_pct": 11.795973,
      "l2_pct": 15.37494,
      "warps_active_pct": 14.14034,
      "waves": 0.61,
      "sm_active_over_elapsed": 0.374298346828143
     }
    }
   ],
   "dominant_kernel": "nvjet_sm90_tst_64x8_64x16_4x1_v_bz_bias_TNT",
   "primary": "partial_wave_tail",
   "primary_rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
   "alternative_explanation": "Small M for this rank shard; multi-kernel split",
   "counter_experiment": "Batch/tile change or larger per-rank M; compare waves and duration",
   "status": "classified"
  },
  {
   "group": "dit:aten.linear.default:{'M': 3, 'N': 5120, 'K': 5120}",
   "resolution": "480x832",
   "stage": "dit",
   "op": "aten.linear.default",
   "shape_parameters": {
    "M": 3,
    "N": 5120,
    "K": 5120
   },
   "input_sha256": "a6cb8d5375b991c9c5d1a1e5674fa86eb61d10648b2ccf71bb7ec40c56f2ac1e",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_capability_batch_dit_rank0_ca590b3a668a9ecf_v1",
   "scheduler_note": null,
   "nominal_mac_flops": 157286400,
   "summed_kernel_us": 17.344,
   "kernels": [
    {
     "kernel": "nvjet_sm90_tst_64x8_64x16_4x1_v_bz_bias_TNT",
     "metrics": {
      "dur_us": 17.344,
      "sm_pct": 8.348938,
      "tensor_pct": 2.470002,
      "dram_pct": 66.960798,
      "l2_pct": 66.232482,
      "warps_active_pct": 13.346335,
      "max_warps_pct": 18.75,
      "waves": 0.61,
      "grid": 80.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 164.308,
      "smem_static_b": 0.0,
      "l2_hit_pct": 2.837203,
      "sm_cycles_active": 16652.856061,
      "cycles_elapsed": 31565.0,
      "dram_read_mb": 52.487168,
      "dram_write_mb": 3.20896,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "dram_bandwidth_bound",
     "rule": "DRAM throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 8.348938,
      "tensor_pct": 2.470002,
      "dram_pct": 66.960798,
      "l2_pct": 66.232482,
      "warps_active_pct": 13.346335,
      "waves": 0.61,
      "sm_active_over_elapsed": 0.5275734535403136
     }
    }
   ],
   "dominant_kernel": "nvjet_sm90_tst_64x8_64x16_4x1_v_bz_bias_TNT",
   "primary": "dram_bandwidth_bound",
   "primary_rule": "DRAM throughput >= 60% of peak sustained",
   "alternative_explanation": "Cache-unfriendly layout or spills inflating traffic",
   "counter_experiment": "Layout/fusion that lowers DRAM bytes; check bytes and time fall together",
   "status": "classified"
  },
  {
   "group": "dit:aten.linear.default:{'M': 3, 'N': 30720, 'K': 5120}",
   "resolution": "480x832",
   "stage": "dit",
   "op": "aten.linear.default",
   "shape_parameters": {
    "M": 3,
    "N": 30720,
    "K": 5120
   },
   "input_sha256": "d6a2e382ecf0fdad711df2cd1bcb239a004321ae666d26aa04920718cf49bf81",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_capability_batch_dit_rank0_22805ed4eb786e66_v1",
   "scheduler_note": null,
   "nominal_mac_flops": 943718400,
   "summed_kernel_us": 79.808,
   "kernels": [
    {
     "kernel": "nvjet_sm90_tst_64x8_64x16_2x1_v_bz_bias_TNT",
     "metrics": {
      "dur_us": 79.808,
      "sm_pct": 10.324124,
      "tensor_pct": 3.269643,
      "dram_pct": 83.319173,
      "l2_pct": 86.567175,
      "warps_active_pct": 14.452215,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 164.308,
      "smem_static_b": 0.0,
      "l2_hit_pct": 4.57719,
      "sm_cycles_active": 127465.310606,
      "cycles_elapsed": 143064.0,
      "dram_read_mb": 314.74688,
      "dram_write_mb": 5.089536,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "dram_bandwidth_bound",
     "rule": "DRAM throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 10.324124,
      "tensor_pct": 3.269643,
      "dram_pct": 83.319173,
      "l2_pct": 86.567175,
      "warps_active_pct": 14.452215,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.8909670539478834
     }
    }
   ],
   "dominant_kernel": "nvjet_sm90_tst_64x8_64x16_2x1_v_bz_bias_TNT",
   "primary": "dram_bandwidth_bound",
   "primary_rule": "DRAM throughput >= 60% of peak sustained",
   "alternative_explanation": "Cache-unfriendly layout or spills inflating traffic",
   "counter_experiment": "Layout/fusion that lowers DRAM bytes; check bytes and time fall together",
   "status": "classified"
  },
  {
   "group": "dit:aten.linear.default:{'M': 1170, 'N': 15360, 'K': 5120}",
   "resolution": "480x832",
   "stage": "dit",
   "op": "aten.linear.default",
   "shape_parameters": {
    "M": 1170,
    "N": 15360,
    "K": 5120
   },
   "input_sha256": "4f1e9988bf86b3f91f274f9b8c3bf33bf42b39945181ce6f09fbce8e0132cb6f",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_capability_batch_dit_rank0_fcebe966800b94b1_v1",
   "scheduler_note": null,
   "nominal_mac_flops": 184025088000,
   "summed_kernel_us": 252.832,
   "kernels": [
    {
     "kernel": "nvjet_sm90_tst_256x152_64x4_1x2_h_bz_coopA_bias_TNT",
     "metrics": {
      "dur_us": 252.832,
      "sm_pct": 85.238963,
      "tensor_pct": 85.238963,
      "dram_pct": 24.405996,
      "l2_pct": 50.066814,
      "warps_active_pct": 14.6146,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 225.476,
      "smem_static_b": 0.0,
      "l2_hit_pct": 62.603473,
      "sm_cycles_active": 370155.863636,
      "cycles_elapsed": 420555.0,
      "dram_read_mb": 262.954496,
      "dram_write_mb": 33.932544,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 85.238963,
      "tensor_pct": 85.238963,
      "dram_pct": 24.405996,
      "l2_pct": 50.066814,
      "warps_active_pct": 14.6146,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.8801604157268372
     }
    }
   ],
   "dominant_kernel": "nvjet_sm90_tst_256x152_64x4_1x2_h_bz_coopA_bias_TNT",
   "primary": "compute_bound",
   "primary_rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
   "alternative_explanation": "Padding/redundant work inside nominal FLOPs; instruction mix vs tensor pipe",
   "counter_experiment": "Same dtype better tile/kernel; compare effective vs executed FLOPs",
   "status": "classified"
  },
  {
   "group": "dit:aten.linear.default:{'M': 1170, 'N': 5120, 'K': 5120}",
   "resolution": "480x832",
   "stage": "dit",
   "op": "aten.linear.default",
   "shape_parameters": {
    "M": 1170,
    "N": 5120,
    "K": 5120
   },
   "input_sha256": "a64f7da6eb1fcc9779d40258d974eb529c2397aaba4df6e32363d7c5d4e73065",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_capability_batch_dit_rank0_10a9e2fea034c6cf_v1",
   "scheduler_note": null,
   "nominal_mac_flops": 61341696000,
   "summed_kernel_us": 87.104,
   "kernels": [
    {
     "kernel": "nvjet_sm90_tst_128x200_64x5_2x1_v_bz_coopA_bias_TNT",
     "metrics": {
      "dur_us": 87.104,
      "sm_pct": 81.160631,
      "tensor_pct": 81.160631,
      "dram_pct": 20.235254,
      "l2_pct": 42.130195,
      "warps_active_pct": 14.427094,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 226.492,
      "smem_static_b": 0.0,
      "l2_hit_pct": 77.359764,
      "sm_cycles_active": 124644.174242,
      "cycles_elapsed": 149125.0,
      "dram_read_mb": 74.980864,
      "dram_write_mb": 9.79072,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 81.160631,
      "tensor_pct": 81.160631,
      "dram_pct": 20.235254,
      "l2_pct": 42.130195,
      "warps_active_pct": 14.427094,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.8358368767275774
     }
    }
   ],
   "dominant_kernel": "nvjet_sm90_tst_128x200_64x5_2x1_v_bz_coopA_bias_TNT",
   "primary": "compute_bound",
   "primary_rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
   "alternative_explanation": "Padding/redundant work inside nominal FLOPs; instruction mix vs tensor pipe",
   "counter_experiment": "Same dtype better tile/kernel; compare effective vs executed FLOPs",
   "status": "classified"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 16, 3, 60, 104], 'weight': [16, 16, 1, 1, 1], 'output': [1, 16, 3, 60, 104], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "480x832",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     16,
     3,
     60,
     104
    ],
    "weight": [
     16,
     16,
     1,
     1,
     1
    ],
    "output": [
     1,
     16,
     3,
     60,
     104
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "32bcd9b4d8be26c876626649d4ff2e623424efe309222bcd7e80afc47310ded4",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_capability_batch_vae_rank0_6638f430ac58a438_v1",
   "scheduler_note": null,
   "nominal_mac_flops": 9584640,
   "summed_kernel_us": 8.511999999999999,
   "kernels": [
    {
     "kernel": "void cutlass::Kernel2<cutlass_80_tensorop_s1688gemm_128x64_16x6_nn_align4>(Params)",
     "metrics": {
      "dur_us": 4.928,
      "sm_pct": 12.442371,
      "tensor_pct": 2.353856,
      "dram_pct": 5.21175,
      "l2_pct": 13.814379,
      "warps_active_pct": 7.051773,
      "max_warps_pct": 18.75,
      "waves": 0.38,
      "grid": 152.0,
      "block": 128.0,
      "regs": 136.0,
      "smem_dyn_kb": 73.728,
      "smem_static_b": 0.0,
      "l2_hit_pct": 57.813092,
      "sm_cycles_active": 5263.075758,
      "cycles_elapsed": 9154.0,
      "dram_read_mb": 1.225728,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 3.0,
      "occ_limit_smem": 3.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 12.442371,
      "tensor_pct": 2.353856,
      "dram_pct": 5.21175,
      "l2_pct": 13.814379,
      "warps_active_pct": 7.051773,
      "waves": 0.38,
      "sm_active_over_elapsed": 0.5749481929211274
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 3.584,
      "sm_pct": 15.235932,
      "tensor_pct": 0.507241,
      "dram_pct": 7.119418,
      "l2_pct": 17.273848,
      "warps_active_pct": 46.267308,
      "max_warps_pct": 100.0,
      "waves": 0.55,
      "grid": 1170.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 53.977594,
      "sm_cycles_active": 3592.515152,
      "cycles_elapsed": 7013.0,
      "dram_read_mb": 1.212672,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 15.235932,
      "tensor_pct": 0.507241,
      "dram_pct": 7.119418,
      "l2_pct": 17.273848,
      "warps_active_pct": 46.267308,
      "waves": 0.55,
      "sm_active_over_elapsed": 0.5122651008127763
     }
    }
   ],
   "dominant_kernel": "void cutlass::Kernel2<cutlass_80_tensorop_s1688gemm_128x64_16x6_nn_align4>(Params)",
   "primary": "partial_wave_tail",
   "primary_rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
   "alternative_explanation": "Small M for this rank shard; multi-kernel split",
   "counter_experiment": "Batch/tile change or larger per-rank M; compare waves and duration",
   "status": "classified"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 16, 3, 62, 28], 'weight': [384, 16, 3, 3, 3], 'output': [1, 384, 1, 60, 26], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "480x832",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     16,
     3,
     62,
     28
    ],
    "weight": [
     384,
     16,
     3,
     3,
     3
    ],
    "output": [
     1,
     384,
     1,
     60,
     26
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "2b9e80cee384305d79045df290612b04bcfe041bd67dc69a2bc096e45a6c5872",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_capability_batch_vae_rank0_29ef146661136bf4_v1",
   "scheduler_note": null,
   "nominal_mac_flops": 517570560,
   "summed_kernel_us": 26.944,
   "kernels": [
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 2>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 2.816,
      "sm_pct": 6.541211,
      "tensor_pct": 0.089078,
      "dram_pct": 2.537262,
      "l2_pct": 7.217338,
      "warps_active_pct": 15.343409,
      "max_warps_pct": 100.0,
      "waves": 0.15,
      "grid": 163.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 61.098724,
      "sm_cycles_active": 2737.757576,
      "cycles_elapsed": 5554.0,
      "dram_read_mb": 0.342528,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 6.541211,
      "tensor_pct": 0.089078,
      "dram_pct": 2.537262,
      "l2_pct": 7.217338,
      "warps_active_pct": 15.343409,
      "waves": 0.15,
      "sm_active_over_elapsed": 0.4929343853078862
     }
    },
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 2>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 3.232,
      "sm_pct": 17.657784,
      "tensor_pct": 0.185842,
      "dram_pct": 4.40506,
      "l2_pct": 11.185747,
      "warps_active_pct": 33.557148,
      "max_warps_pct": 100.0,
      "waves": 0.36,
      "grid": 384.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 55.920487,
      "sm_cycles_active": 3194.05303,
      "cycles_elapsed": 6282.0,
      "dram_read_mb": 0.672768,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 17.657784,
      "tensor_pct": 0.185842,
      "dram_pct": 4.40506,
      "l2_pct": 11.185747,
      "warps_active_pct": 33.557148,
      "waves": 0.36,
      "sm_active_over_elapsed": 0.5084452451448583
     }
    },
    {
     "kernel": "sm80_xmma_fprop_implicit_gemm_tf32f32_tf32f32_f32_nhwckrsc_nchw_tilesize128x64x32_stage5_warpsize2x2x1_g1_tensor16x8x8_e",
     "metrics": {
      "dur_us": 16.32,
      "sm_pct": 20.669159,
      "tensor_pct": 20.669159,
      "dram_pct": 1.351313,
      "l2_pct": 16.424589,
      "warps_active_pct": 6.243267,
      "max_warps_pct": 6.25,
      "waves": 0.59,
      "grid": 78.0,
      "block": 128.0,
      "regs": 168.0,
      "smem_dyn_kb": 122.88,
      "smem_static_b": 0.0,
      "l2_hit_pct": 94.638615,
      "sm_cycles_active": 15331.212121,
      "cycles_elapsed": 29933.0,
      "dram_read_mb": 1.059328,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 3.0,
      "occ_limit_smem": 1.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 20.669159,
      "tensor_pct": 20.669159,
      "dram_pct": 1.351313,
      "l2_pct": 16.424589,
      "warps_active_pct": 6.243267,
      "waves": 0.59,
      "sm_active_over_elapsed": 0.5121842822637224
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 4.576,
      "sm_pct": 22.863568,
      "tensor_pct": 0.791984,
      "dram_pct": 11.044218,
      "l2_pct": 25.407107,
      "warps_active_pct": 71.799906,
      "max_warps_pct": 100.0,
      "waves": 1.11,
      "grid": 2340.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 53.905335,
      "sm_cycles_active": 4784.409091,
      "cycles_elapsed": 8978.0,
      "dram_read_mb": 2.412288,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 22.863568,
      "tensor_pct": 0.791984,
      "dram_pct": 11.044218,
      "l2_pct": 25.407107,
      "warps_active_pct": 71.799906,
      "waves": 1.11,
      "sm_active_over_elapsed": 0.5329036635108042
     }
    }
   ],
   "dominant_kernel": "sm80_xmma_fprop_implicit_gemm_tf32f32_tf32f32_f32_nhwckrsc_nchw_tilesize128x64x32_stage5_warpsize2x2x1_g1_tensor16x8x8_e",
   "primary": "partial_wave_tail",
   "primary_rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
   "alternative_explanation": "Small M for this rank shard; multi-kernel split",
   "counter_experiment": "Batch/tile change or larger per-rank M; compare waves and duration",
   "status": "classified"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 384, 3, 62, 28], 'weight': [384, 384, 3, 3, 3], 'output': [1, 384, 1, 60, 26], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "480x832",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     384,
     3,
     62,
     28
    ],
    "weight": [
     384,
     384,
     3,
     3,
     3
    ],
    "output": [
     1,
     384,
     1,
     60,
     26
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "6309c12f24013edabd88edced7c2ce22672322d203206b12ba218c88696effb5",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_capability_batch_vae_rank0_551fa1ff3095f40c_v1",
   "scheduler_note": null,
   "nominal_mac_flops": 12421693440,
   "summed_kernel_us": 122.848,
   "kernels": [
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 6.848,
      "sm_pct": 27.404928,
      "tensor_pct": 0.0,
      "dram_pct": 24.416844,
      "l2_pct": 53.461993,
      "warps_active_pct": 73.664364,
      "max_warps_pct": 100.0,
      "waves": 1.85,
      "grid": 1956.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 56.459204,
      "sm_cycles_active": 8734.734848,
      "cycles_elapsed": 13493.0,
      "dram_read_mb": 8.008448,
      "dram_write_mb": 0.002048,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 27.404928,
      "tensor_pct": 0.0,
      "dram_pct": 24.416844,
      "l2_pct": 53.461993,
      "warps_active_pct": 73.664364,
      "waves": 1.85,
      "sm_active_over_elapsed": 0.6473530606981398
     }
    },
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 9.984,
      "sm_pct": 63.899285,
      "tensor_pct": 0.0,
      "dram_pct": 34.966809,
      "l2_pct": 69.209084,
      "warps_active_pct": 84.724705,
      "max_warps_pct": 100.0,
      "waves": 4.36,
      "grid": 4608.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 56.383965,
      "sm_cycles_active": 15370.712121,
      "cycles_elapsed": 19576.0,
      "dram_read_mb": 15.93472,
      "dram_write_mb": 0.804352,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 63.899285,
      "tensor_pct": 0.0,
      "dram_pct": 34.966809,
      "l2_pct": 69.209084,
      "warps_active_pct": 84.724705,
      "waves": 4.36,
      "sm_active_over_elapsed": 0.7851814528504292
     }
    },
    {
     "kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize64x64x32_warpgroupsize1x1x1_g1_execute_segment_k_",
     "metrics": {
      "dur_us": 97.824,
      "sm_pct": 37.579717,
      "tensor_pct": 37.579717,
      "dram_pct": 5.36294,
      "l2_pct": 41.273313,
      "warps_active_pct": 13.550294,
      "max_warps_pct": 18.75,
      "waves": 0.91,
      "grid": 120.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 231.424,
      "smem_static_b": 0.0,
      "l2_hit_pct": 89.627586,
      "sm_cycles_active": 118053.606061,
      "cycles_elapsed": 177046.0,
      "dram_read_mb": 23.975168,
      "dram_write_mb": 1.262592,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 37.579717,
      "tensor_pct": 37.579717,
      "dram_pct": 5.36294,
      "l2_pct": 41.273313,
      "warps_active_pct": 13.550294,
      "waves": 0.91,
      "sm_active_over_elapsed": 0.666796234091705
     }
    },
    {
     "kernel": "void cudnn::nhwcToNchwKernel<float, float, float, 1, 0, 0>(cudnn::nhwc2nchw_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 3.648,
      "sm_pct": 14.880553,
      "tensor_pct": 0.0,
      "dram_pct": 13.778107,
      "l2_pct": 34.779502,
      "warps_active_pct": 51.828874,
      "max_warps_pct": 100.0,
      "waves": 0.56,
      "grid": 588.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 55.606173,
      "sm_cycles_active": 3581.128788,
      "cycles_elapsed": 7205.0,
      "dram_read_mb": 2.41152,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 14.880553,
      "tensor_pct": 0.0,
      "dram_pct": 13.778107,
      "l2_pct": 34.779502,
      "warps_active_pct": 51.828874,
      "waves": 0.56,
      "sm_active_over_elapsed": 0.49703383594725886
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 4.544,
      "sm_pct": 23.116983,
      "tensor_pct": 0.801721,
      "dram_pct": 11.183862,
      "l2_pct": 25.764575,
      "warps_active_pct": 70.343863,
      "max_warps_pct": 100.0,
      "waves": 1.11,
      "grid": 2340.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 53.961524,
      "sm_cycles_active": 4745.681818,
      "cycles_elapsed": 8878.0,
      "dram_read_mb": 2.412288,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 23.116983,
      "tensor_pct": 0.801721,
      "dram_pct": 11.183862,
      "l2_pct": 25.764575,
      "warps_active_pct": 70.343863,
      "waves": 1.11,
      "sm_active_over_elapsed": 0.5345440209506646
     }
    }
   ],
   "dominant_kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize64x64x32_warpgroupsize1x1x1_g1_execute_segment_k_",
   "primary": "partial_wave_tail",
   "primary_rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
   "alternative_explanation": "Small M for this rank shard; multi-kernel split",
   "counter_experiment": "Batch/tile change or larger per-rank M; compare waves and duration",
   "status": "classified"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 192, 1, 120, 52], 'weight': [384, 192, 1, 1, 1], 'output': [1, 384, 1, 120, 52], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "480x832",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     192,
     1,
     120,
     52
    ],
    "weight": [
     384,
     192,
     1,
     1,
     1
    ],
    "output": [
     1,
     384,
     1,
     120,
     52
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "1e43418295ca6422f27afcf0b9940aa818fbb12fcf1013c12dfcf7a60f86e7d7",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_capability_batch_vae_rank0_c16d11a7624e1a8c_v1",
   "scheduler_note": null,
   "nominal_mac_flops": 920125440,
   "summed_kernel_us": 21.152,
   "kernels": [
    {
     "kernel": "void cutlass::Kernel2<cutlass_80_tensorop_s1688gemm_64x128_32x3_nn_align4>(Params)",
     "metrics": {
      "dur_us": 12.384,
      "sm_pct": 25.920931,
      "tensor_pct": 23.075477,
      "dram_pct": 8.607037,
      "l2_pct": 33.983188,
      "warps_active_pct": 13.551422,
      "max_warps_pct": 18.75,
      "waves": 0.79,
      "grid": 312.0,
      "block": 128.0,
      "regs": 168.0,
      "smem_dyn_kb": 73.728,
      "smem_static_b": 0.0,
      "l2_hit_pct": 80.711755,
      "sm_cycles_active": 14789.969697,
      "cycles_elapsed": 22571.0,
      "dram_read_mb": 5.116928,
      "dram_write_mb": 3.072,
      "occ_limit_regs": 3.0,
      "occ_limit_smem": 3.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 25.920931,
      "tensor_pct": 23.075477,
      "dram_pct": 8.607037,
      "l2_pct": 33.983188,
      "warps_active_pct": 13.551422,
      "waves": 0.79,
      "sm_active_over_elapsed": 0.6552642637455142
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 8.768,
      "sm_pct": 46.554328,
      "tensor_pct": 1.660704,
      "dram_pct": 22.911033,
      "l2_pct": 51.457793,
      "warps_active_pct": 74.651981,
      "max_warps_pct": 100.0,
      "waves": 4.43,
      "grid": 9360.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 52.470894,
      "sm_cycles_active": 13692.030303,
      "cycles_elapsed": 17205.0,
      "dram_read_mb": 9.600512,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 46.554328,
      "tensor_pct": 1.660704,
      "dram_pct": 22.911033,
      "l2_pct": 51.457793,
      "warps_active_pct": 74.651981,
      "waves": 4.43,
      "sm_active_over_elapsed": 0.795816931299041
     }
    }
   ],
   "dominant_kernel": "void cutlass::Kernel2<cutlass_80_tensorop_s1688gemm_64x128_32x3_nn_align4>(Params)",
   "primary": "partial_wave_tail",
   "primary_rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
   "alternative_explanation": "Small M for this rank shard; multi-kernel split",
   "counter_experiment": "Batch/tile change or larger per-rank M; compare waves and duration",
   "status": "classified"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 192, 3, 122, 54], 'weight': [384, 192, 3, 3, 3], 'output': [1, 384, 1, 120, 52], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "480x832",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     192,
     3,
     122,
     54
    ],
    "weight": [
     384,
     192,
     3,
     3,
     3
    ],
    "output": [
     1,
     384,
     1,
     120,
     52
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "3193a32ddc024e9d34eb5446f9909408dde38d963ebc5cb9654998ecf661a820",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_capability_batch_vae_rank0_081e405bd1e07dce_v1",
   "scheduler_note": null,
   "nominal_mac_flops": 24843386880,
   "summed_kernel_us": 158.56,
   "kernels": [
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 9.568,
      "sm_pct": 36.450718,
      "tensor_pct": 0.0,
      "dram_pct": 34.985392,
      "l2_pct": 70.74527,
      "warps_active_pct": 82.59028,
      "max_warps_pct": 100.0,
      "waves": 3.51,
      "grid": 3708.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 57.439349,
      "sm_cycles_active": 14773.583333,
      "cycles_elapsed": 18859.0,
      "dram_read_mb": 15.187968,
      "dram_write_mb": 0.875264,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "l2_bandwidth_bound",
     "rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
     "evidence": {
      "sm_pct": 36.450718,
      "tensor_pct": 0.0,
      "dram_pct": 34.985392,
      "l2_pct": 70.74527,
      "warps_active_pct": 82.59028,
      "waves": 3.51,
      "sm_active_over_elapsed": 0.7833704508722626
     }
    },
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 6.624,
      "sm_pct": 48.401978,
      "tensor_pct": 0.0,
      "dram_pct": 25.123701,
      "l2_pct": 55.958628,
      "warps_active_pct": 80.361473,
      "max_warps_pct": 100.0,
      "waves": 2.18,
      "grid": 2304.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 56.181959,
      "sm_cycles_active": 8730.106061,
      "cycles_elapsed": 13007.0,
      "dram_read_mb": 7.971584,
      "dram_write_mb": 0.000768,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 48.401978,
      "tensor_pct": 0.0,
      "dram_pct": 25.123701,
      "l2_pct": 55.958628,
      "warps_active_pct": 80.361473,
      "waves": 2.18,
      "sm_active_over_elapsed": 0.6711852126547244
     }
    },
    {
     "kernel": "void cask_plugin__5x_cudnn::init_device_workspace_kernel<xmma__5x_cudnn::Warp_specialized_params_non_template<xmma__5x_c",
     "metrics": {
      "dur_us": 3.68,
      "sm_pct": 0.028338,
      "tensor_pct": 0.0,
      "dram_pct": 0.026364,
      "l2_pct": 0.302623,
      "warps_active_pct": 1.519546,
      "max_warps_pct": 50.0,
      "waves": 0.0,
      "grid": 1.0,
      "block": 1.0,
      "regs": 16.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 47.312842,
      "sm_cycles_active": 25.628788,
      "cycles_elapsed": 7195.0,
      "dram_read_mb": 0.004608,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 128.0,
      "occ_limit_smem": 32.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 0.028338,
      "tensor_pct": 0.0,
      "dram_pct": 0.026364,
      "l2_pct": 0.302623,
      "warps_active_pct": 1.519546,
      "waves": 0.0,
      "sm_active_over_elapsed": 0.0035620275191104935
     }
    },
    {
     "kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
     "metrics": {
      "dur_us": 122.688,
      "sm_pct": 56.462241,
      "tensor_pct": 56.462241,
      "dram_pct": 5.56358,
      "l2_pct": 51.396564,
      "warps_active_pct": 14.433255,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 231.424,
      "smem_static_b": 0.0,
      "l2_hit_pct": 93.811249,
      "sm_cycles_active": 183305.69697,
      "cycles_elapsed": 220290.0,
      "dram_read_mb": 24.830208,
      "dram_write_mb": 8.004352,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 56.462241,
      "tensor_pct": 56.462241,
      "dram_pct": 5.56358,
      "l2_pct": 51.396564,
      "warps_active_pct": 14.433255,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.8321108401198419
     }
    },
    {
     "kernel": "void cudnn::nhwcToNchwKernel<float, float, float, 1, 0, 0>(cudnn::nhwc2nchw_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 7.2,
      "sm_pct": 27.470815,
      "tensor_pct": 0.0,
      "dram_pct": 27.859537,
      "l2_pct": 58.099129,
      "warps_active_pct": 81.932699,
      "max_warps_pct": 100.0,
      "waves": 2.22,
      "grid": 2340.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 52.341911,
      "sm_cycles_active": 9856.674242,
      "cycles_elapsed": 14181.0,
      "dram_read_mb": 9.6,
      "dram_write_mb": 0.018176,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 27.470815,
      "tensor_pct": 0.0,
      "dram_pct": 27.859537,
      "l2_pct": 58.099129,
      "warps_active_pct": 81.932699,
      "waves": 2.22,
      "sm_active_over_elapsed": 0.6950620014103377
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 8.8,
      "sm_pct": 46.141705,
      "tensor_pct": 1.647252,
      "dram_pct": 22.736937,
      "l2_pct": 50.112379,
      "warps_active_pct": 73.619081,
      "max_warps_pct": 100.0,
      "waves": 4.43,
      "grid": 9360.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 52.725209,
      "sm_cycles_active": 13598.060606,
      "cycles_elapsed": 17345.0,
      "dram_read_mb": 9.600512,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 46.141705,
      "tensor_pct": 1.647252,
      "dram_pct": 22.736937,
      "l2_pct": 50.112379,
      "warps_active_pct": 73.619081,
      "waves": 4.43,
      "sm_active_over_elapsed": 0.7839758204669933
     }
    }
   ],
   "dominant_kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
   "primary": "partial_wave_tail",
   "primary_rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
   "alternative_explanation": "Small M for this rank shard; multi-kernel split",
   "counter_experiment": "Batch/tile change or larger per-rank M; compare waves and duration",
   "status": "classified"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 384, 3, 122, 54], 'weight': [384, 384, 3, 3, 3], 'output': [1, 384, 1, 120, 52], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "480x832",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     384,
     3,
     122,
     54
    ],
    "weight": [
     384,
     384,
     3,
     3,
     3
    ],
    "output": [
     1,
     384,
     1,
     120,
     52
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "c7f10fe8b8ca84229e785645c70c003456c3950c79cdd176694e5bc86c9247f4",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_capability_batch_vae_rank0_8601bc3a88374d99_v1",
   "scheduler_note": null,
   "nominal_mac_flops": 49686773760,
   "summed_kernel_us": 287.104,
   "kernels": [
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 16.64,
      "sm_pct": 41.695108,
      "tensor_pct": 0.0,
      "dram_pct": 52.385208,
      "l2_pct": 79.417837,
      "warps_active_pct": 86.940903,
      "max_warps_pct": 100.0,
      "waves": 7.02,
      "grid": 7416.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 57.090786,
      "sm_cycles_active": 28106.666667,
      "cycles_elapsed": 32841.0,
      "dram_read_mb": 30.366976,
      "dram_write_mb": 11.497472,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "l2_bandwidth_bound",
     "rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
     "evidence": {
      "sm_pct": 41.695108,
      "tensor_pct": 0.0,
      "dram_pct": 52.385208,
      "l2_pct": 79.417837,
      "warps_active_pct": 86.940903,
      "waves": 7.02,
      "sm_active_over_elapsed": 0.8558407681556591
     }
    },
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 10.144,
      "sm_pct": 62.70737,
      "tensor_pct": 0.0,
      "dram_pct": 34.251154,
      "l2_pct": 69.326632,
      "warps_active_pct": 83.711812,
      "max_warps_pct": 100.0,
      "waves": 4.36,
      "grid": 4608.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 56.051816,
      "sm_cycles_active": 15747.295455,
      "cycles_elapsed": 19924.0,
      "dram_read_mb": 15.934464,
      "dram_write_mb": 0.766976,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 62.70737,
      "tensor_pct": 0.0,
      "dram_pct": 34.251154,
      "l2_pct": 69.326632,
      "warps_active_pct": 83.711812,
      "waves": 4.36,
      "sm_active_over_elapsed": 0.7903681718028508
     }
    },
    {
     "kernel": "void cask_plugin__5x_cudnn::init_device_workspace_kernel<xmma__5x_cudnn::Warp_specialized_params_non_template<xmma__5x_c",
     "metrics": {
      "dur_us": 3.776,
      "sm_pct": 0.027533,
      "tensor_pct": 0.0,
      "dram_pct": 0.027022,
      "l2_pct": 0.303405,
      "warps_active_pct": 1.552996,
      "max_warps_pct": 50.0,
      "waves": 0.0,
      "grid": 1.0,
      "block": 1.0,
      "regs": 16.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 47.369421,
      "sm_cycles_active": 24.909091,
      "cycles_elapsed": 7409.0,
      "dram_read_mb": 0.004864,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 128.0,
      "occ_limit_smem": 32.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 0.027533,
      "tensor_pct": 0.0,
      "dram_pct": 0.027022,
      "l2_pct": 0.303405,
      "warps_active_pct": 1.552996,
      "waves": 0.0,
      "sm_active_over_elapsed": 0.003362004454042381
     }
    },
    {
     "kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
     "metrics": {
      "dur_us": 240.352,
      "sm_pct": 57.441182,
      "tensor_pct": 57.441182,
      "dram_pct": 7.155502,
      "l2_pct": 52.670803,
      "warps_active_pct": 13.956357,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 231.424,
      "smem_static_b": 0.0,
      "l2_hit_pct": 91.132285,
      "sm_cycles_active": 352163.5,
      "cycles_elapsed": 432053.0,
      "dram_read_mb": 68.61952,
      "dram_write_mb": 14.121216,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 57.441182,
      "tensor_pct": 57.441182,
      "dram_pct": 7.155502,
      "l2_pct": 52.670803,
      "warps_active_pct": 13.956357,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.8150932871661578
     }
    },
    {
     "kernel": "void cudnn::nhwcToNchwKernel<float, float, float, 1, 0, 0>(cudnn::nhwc2nchw_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 7.392,
      "sm_pct": 26.789002,
      "tensor_pct": 0.0,
      "dram_pct": 27.118987,
      "l2_pct": 57.452752,
      "warps_active_pct": 80.668169,
      "max_warps_pct": 100.0,
      "waves": 2.22,
      "grid": 2340.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 52.281985,
      "sm_cycles_active": 9706.575758,
      "cycles_elapsed": 14547.0,
      "dram_read_mb": 9.6,
      "dram_write_mb": 0.006144,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 26.789002,
      "tensor_pct": 0.0,
      "dram_pct": 27.118987,
      "l2_pct": 57.452752,
      "warps_active_pct": 80.668169,
      "waves": 2.22,
      "sm_active_over_elapsed": 0.6672561873925896
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 8.8,
      "sm_pct": 46.241571,
      "tensor_pct": 1.6498,
      "dram_pct": 22.771544,
      "l2_pct": 50.164747,
      "warps_active_pct": 70.432619,
      "max_warps_pct": 100.0,
      "waves": 4.43,
      "grid": 9360.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 52.462156,
      "sm_cycles_active": 14371.575758,
      "cycles_elapsed": 17312.0,
      "dram_read_mb": 9.600512,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 46.241571,
      "tensor_pct": 1.6498,
      "dram_pct": 22.771544,
      "l2_pct": 50.164747,
      "warps_active_pct": 70.432619,
      "waves": 4.43,
      "sm_active_over_elapsed": 0.8301510950785582
     }
    }
   ],
   "dominant_kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
   "primary": "partial_wave_tail",
   "primary_rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
   "alternative_explanation": "Small M for this rank shard; multi-kernel split",
   "counter_experiment": "Batch/tile change or larger per-rank M; compare waves and duration",
   "status": "classified"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 192, 3, 242, 106], 'weight': [192, 192, 3, 3, 3], 'output': [1, 192, 1, 240, 104], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "480x832",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     192,
     3,
     242,
     106
    ],
    "weight": [
     192,
     192,
     3,
     3,
     3
    ],
    "output": [
     1,
     192,
     1,
     240,
     104
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "0cc0a21ce474602c53c3e5621e64598ed7be5a2c16603fc968441df41de71330",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_capability_batch_vae_rank0_99475e040d65d084_v1",
   "scheduler_note": null,
   "nominal_mac_flops": 49686773760,
   "summed_kernel_us": 230.72,
   "kernels": [
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 34.08,
      "sm_pct": 39.224356,
      "tensor_pct": 0.0,
      "dram_pct": 60.726759,
      "l2_pct": 81.305041,
      "warps_active_pct": 90.207582,
      "max_warps_pct": 100.0,
      "waves": 13.66,
      "grid": 14430.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 56.436241,
      "sm_cycles_active": 61442.159091,
      "cycles_elapsed": 67361.0,
      "dram_read_mb": 59.112448,
      "dram_write_mb": 40.355328,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "dram_bandwidth_bound",
     "rule": "DRAM throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 39.224356,
      "tensor_pct": 0.0,
      "dram_pct": 60.726759,
      "l2_pct": 81.305041,
      "warps_active_pct": 90.207582,
      "waves": 13.66,
      "sm_active_over_elapsed": 0.9121325261056101
     }
    },
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 4.896,
      "sm_pct": 33.361319,
      "tensor_pct": 0.0,
      "dram_pct": 17.081921,
      "l2_pct": 40.884687,
      "warps_active_pct": 73.72203,
      "max_warps_pct": 100.0,
      "waves": 1.09,
      "grid": 1152.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 57.925149,
      "sm_cycles_active": 5200.878788,
      "cycles_elapsed": 9592.0,
      "dram_read_mb": 3.990016,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 33.361319,
      "tensor_pct": 0.0,
      "dram_pct": 17.081921,
      "l2_pct": 40.884687,
      "warps_active_pct": 73.72203,
      "waves": 1.09,
      "sm_active_over_elapsed": 0.5422100487906589
     }
    },
    {
     "kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
     "metrics": {
      "dur_us": 166.176,
      "sm_pct": 85.708077,
      "tensor_pct": 85.708077,
      "dram_pct": 9.421867,
      "l2_pct": 68.827558,
      "warps_active_pct": 14.85885,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 231.424,
      "smem_static_b": 0.0,
      "l2_hit_pct": 89.957138,
      "sm_cycles_active": 277166.030303,
      "cycles_elapsed": 289400.0,
      "dram_read_mb": 64.303104,
      "dram_write_mb": 11.027968,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 85.708077,
      "tensor_pct": 85.708077,
      "dram_pct": 9.421867,
      "l2_pct": 68.827558,
      "warps_active_pct": 14.85885,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.957726435048376
     }
    },
    {
     "kernel": "void cudnn::nhwcToNchwKernel<float, float, float, 1, 0, 0>(cudnn::nhwc2nchw_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 10.752,
      "sm_pct": 37.000924,
      "tensor_pct": 0.0,
      "dram_pct": 41.05373,
      "l2_pct": 73.968927,
      "warps_active_pct": 83.475165,
      "max_warps_pct": 100.0,
      "waves": 4.43,
      "grid": 4680.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 51.687049,
      "sm_cycles_active": 17533.234848,
      "cycles_elapsed": 21155.0,
      "dram_read_mb": 19.1872,
      "dram_write_mb": 1.970688,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "l2_bandwidth_bound",
     "rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
     "evidence": {
      "sm_pct": 37.000924,
      "tensor_pct": 0.0,
      "dram_pct": 41.05373,
      "l2_pct": 73.968927,
      "warps_active_pct": 83.475165,
      "waves": 4.43,
      "sm_active_over_elapsed": 0.8287986219806193
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 14.816,
      "sm_pct": 54.648822,
      "tensor_pct": 1.959316,
      "dram_pct": 27.255233,
      "l2_pct": 58.75947,
      "warps_active_pct": 77.381408,
      "max_warps_pct": 100.0,
      "waves": 8.86,
      "grid": 18720.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 52.067691,
      "sm_cycles_active": 24793.492424,
      "cycles_elapsed": 29174.0,
      "dram_read_mb": 19.18464,
      "dram_write_mb": 0.17664,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 54.648822,
      "tensor_pct": 1.959316,
      "dram_pct": 27.255233,
      "l2_pct": 58.75947,
      "warps_active_pct": 77.381408,
      "waves": 8.86,
      "sm_active_over_elapsed": 0.8498489210941249
     }
    }
   ],
   "dominant_kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
   "primary": "compute_bound",
   "primary_rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
   "alternative_explanation": "Padding/redundant work inside nominal FLOPs; instruction mix vs tensor pipe",
   "counter_experiment": "Same dtype better tile/kernel; compare effective vs executed FLOPs",
   "status": "classified"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 96, 3, 482, 210], 'weight': [96, 96, 3, 3, 3], 'output': [1, 96, 1, 480, 208], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "480x832",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     96,
     3,
     482,
     210
    ],
    "weight": [
     96,
     96,
     3,
     3,
     3
    ],
    "output": [
     1,
     96,
     1,
     480,
     208
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "1e47965836d719f164296b5c8d52cac8c22a10d9ff8c5f4fdfa48498909f8856",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_capability_vae_v1",
   "scheduler_note": null,
   "nominal_mac_flops": 49686773760,
   "summed_kernel_us": 466.624,
   "kernels": [
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 70.336,
      "sm_pct": 37.540406,
      "tensor_pct": 0.0,
      "dram_pct": 63.710272,
      "l2_pct": 80.815686,
      "warps_active_pct": 93.308252,
      "max_warps_pct": 100.0,
      "waves": 26.96,
      "grid": 28470.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 56.725729,
      "sm_cycles_active": 132556.393939,
      "cycles_elapsed": 138367.0,
      "dram_read_mb": 116.620544,
      "dram_write_mb": 98.860032,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "dram_bandwidth_bound",
     "rule": "DRAM throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 37.540406,
      "tensor_pct": 0.0,
      "dram_pct": 63.710272,
      "l2_pct": 80.815686,
      "warps_active_pct": 93.308252,
      "waves": 26.96,
      "sm_active_over_elapsed": 0.9580058391018089
     }
    },
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 3.52,
      "sm_pct": 12.390404,
      "tensor_pct": 0.0,
      "dram_pct": 5.953501,
      "l2_pct": 15.414774,
      "warps_active_pct": 24.41718,
      "max_warps_pct": 100.0,
      "waves": 0.27,
      "grid": 288.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 58.631017,
      "sm_cycles_active": 3023.19697,
      "cycles_elapsed": 6920.0,
      "dram_read_mb": 1.004288,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 12.390404,
      "tensor_pct": 0.0,
      "dram_pct": 5.953501,
      "l2_pct": 15.414774,
      "warps_active_pct": 24.41718,
      "waves": 0.27,
      "sm_active_over_elapsed": 0.4368781748554913
     }
    },
    {
     "kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize256x32x32_warpgroupsize1x1x1_g1_execute_segment_k",
     "metrics": {
      "dur_us": 345.6,
      "sm_pct": 29.68631,
      "tensor_pct": 29.68631,
      "dram_pct": 8.985101,
      "l2_pct": 60.273264,
      "warps_active_pct": 17.755377,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 231.424,
      "smem_static_b": 0.0,
      "l2_hit_pct": 89.426232,
      "sm_cycles_active": 591719.492424,
      "cycles_elapsed": 620733.0,
      "dram_read_mb": 117.670144,
      "dram_write_mb": 31.733248,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "l2_bandwidth_bound",
     "rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
     "evidence": {
      "sm_pct": 29.68631,
      "tensor_pct": 29.68631,
      "dram_pct": 8.985101,
      "l2_pct": 60.273264,
      "warps_active_pct": 17.755377,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.9532592796323057
     }
    },
    {
     "kernel": "void cudnn::nhwcToNchwKernel<float, float, float, 1, 0, 0>(cudnn::nhwc2nchw_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 19.904,
      "sm_pct": 41.754029,
      "tensor_pct": 0.0,
      "dram_pct": 58.781261,
      "l2_pct": 81.369093,
      "warps_active_pct": 89.587993,
      "max_warps_pct": 100.0,
      "waves": 8.86,
      "grid": 9360.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 50.983393,
      "sm_cycles_active": 33855.530303,
      "cycles_elapsed": 38959.0,
      "dram_read_mb": 38.355968,
      "dram_write_mb": 17.823744,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "l2_bandwidth_bound",
     "rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
     "evidence": {
      "sm_pct": 41.754029,
      "tensor_pct": 0.0,
      "dram_pct": 58.781261,
      "l2_pct": 81.369093,
      "warps_active_pct": 89.587993,
      "waves": 8.86,
      "sm_active_over_elapsed": 0.8690040889909905
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 27.264,
      "sm_pct": 59.158799,
      "tensor_pct": 2.126489,
      "dram_pct": 39.056992,
      "l2_pct": 60.717071,
      "warps_active_pct": 79.24734,
      "max_warps_pct": 100.0,
      "waves": 17.73,
      "grid": 37440.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 50.828045,
      "sm_cycles_active": 49380.666667,
      "cycles_elapsed": 53506.0,
      "dram_read_mb": 38.354688,
      "dram_write_mb": 12.763136,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "l2_bandwidth_bound",
     "rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
     "evidence": {
      "sm_pct": 59.158799,
      "tensor_pct": 2.126489,
      "dram_pct": 39.056992,
      "l2_pct": 60.717071,
      "warps_active_pct": 79.24734,
      "waves": 17.73,
      "sm_active_over_elapsed": 0.9228996125107464
     }
    }
   ],
   "dominant_kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize256x32x32_warpgroupsize1x1x1_g1_execute_segment_k",
   "primary": "l2_bandwidth_bound",
   "primary_rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
   "alternative_explanation": "Working set thrashing L2; DRAM counter under-reads write-back",
   "counter_experiment": "Blocking/reuse change; compare lts bytes",
   "status": "classified"
  },
  {
   "group": "output:aten.to.dtype:{}",
   "resolution": "480x832",
   "stage": "output",
   "op": "aten.to.dtype",
   "shape_parameters": {},
   "input_sha256": "897dd2bd3ad5fe7258ee5c3ae9cfd93da4e5e94644902267f0b3d8bd156f63b0",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_capability_batch_output_rank0_f18eb91117dda1c8_v1",
   "scheduler_note": null,
   "nominal_mac_flops": null,
   "summed_kernel_us": 29.824,
   "kernels": [
    {
     "kernel": "void unrolled_elementwise_kernel<direct_copy_kernel_cuda(TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() l",
     "metrics": {
      "dur_us": 29.824,
      "sm_pct": 55.696,
      "tensor_pct": 0.0,
      "dram_pct": 33.813727,
      "l2_pct": 43.414546,
      "warps_active_pct": 86.444578,
      "max_warps_pct": 100.0,
      "waves": 9.97,
      "grid": 21060.0,
      "block": 128.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 21.577663,
      "sm_cycles_active": 54446.05303,
      "cycles_elapsed": 59003.0,
      "dram_read_mb": 43.166464,
      "dram_write_mb": 5.343232,
      "occ_limit_regs": 16.0,
      "occ_limit_smem": 32.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 55.696,
      "tensor_pct": 0.0,
      "dram_pct": 33.813727,
      "l2_pct": 43.414546,
      "warps_active_pct": 86.444578,
      "waves": 9.97,
      "sm_active_over_elapsed": 0.9227675377523177
     }
    }
   ],
   "dominant_kernel": "void unrolled_elementwise_kernel<direct_copy_kernel_cuda(TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() l",
   "primary": "unresolved_no_dominant_limiter",
   "primary_rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
   "alternative_explanation": "Mixed limiter or measurement window artifact",
   "counter_experiment": "Collect stall reasons (smsp__warp_issue_stalled_*) and l1tex/lts bytes",
   "status": "unresolved"
  },
  {
   "group": "output:aten.to.dtype:{}",
   "resolution": "480x832",
   "stage": "output",
   "op": "aten.to.dtype",
   "shape_parameters": {},
   "input_sha256": "a9aab61b63e7cab483e81582c9af41d89d525814ab14de753e8a7b16de6c4f1b",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_capability_output_v1",
   "scheduler_note": null,
   "nominal_mac_flops": null,
   "summed_kernel_us": 39.583999999999996,
   "kernels": [
    {
     "kernel": "void unrolled_elementwise_kernel<direct_copy_kernel_cuda(TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() l",
     "metrics": {
      "dur_us": 39.584,
      "sm_pct": 55.888331,
      "tensor_pct": 0.0,
      "dram_pct": 36.344303,
      "l2_pct": 44.802352,
      "warps_active_pct": 89.899207,
      "max_warps_pct": 100.0,
      "waves": 13.3,
      "grid": 28080.0,
      "block": 128.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 21.34539,
      "sm_cycles_active": 71810.371212,
      "cycles_elapsed": 78244.0,
      "dram_read_mb": 57.544448,
      "dram_write_mb": 11.594752,
      "occ_limit_regs": 16.0,
      "occ_limit_smem": 32.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 55.888331,
      "tensor_pct": 0.0,
      "dram_pct": 36.344303,
      "l2_pct": 44.802352,
      "warps_active_pct": 89.899207,
      "waves": 13.3,
      "sm_active_over_elapsed": 0.9177747969428965
     }
    }
   ],
   "dominant_kernel": "void unrolled_elementwise_kernel<direct_copy_kernel_cuda(TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() l",
   "primary": "unresolved_no_dominant_limiter",
   "primary_rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
   "alternative_explanation": "Mixed limiter or measurement window artifact",
   "counter_experiment": "Collect stall reasons (smsp__warp_issue_stalled_*) and l1tex/lts bytes",
   "status": "unresolved"
  },
  {
   "group": "dit:aten.linear.default:{'M': 1170, 'N': 13824, 'K': 5120}",
   "resolution": "480x832",
   "stage": "dit",
   "op": "aten.linear.default",
   "shape_parameters": {
    "M": 1170,
    "N": 13824,
    "K": 5120
   },
   "input_sha256": "b8c205d5140f4a6d1aabaa7d0946d8eb3aa5bdc6007bff3e8efd26fcf1740c87",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing_80ac9ebe75667de4_v1",
   "scheduler_note": null,
   "nominal_mac_flops": 165622579200,
   "summed_kernel_us": 215.52,
   "kernels": [
    {
     "kernel": "nvjet_sm90_tst_128x200_64x5_2x1_v_bz_coopA_bias_TNT",
     "metrics": {
      "dur_us": 215.52,
      "sm_pct": 91.537864,
      "tensor_pct": 91.537864,
      "dram_pct": 22.141116,
      "l2_pct": 53.502631,
      "warps_active_pct": 14.666908,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 226.492,
      "smem_static_b": 0.0,
      "l2_hit_pct": 73.617604,
      "sm_cycles_active": 329920.545455,
      "cycles_elapsed": 348828.0,
      "dram_read_mb": 199.619072,
      "dram_write_mb": 29.967872,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 91.537864,
      "tensor_pct": 91.537864,
      "dram_pct": 22.141116,
      "l2_pct": 53.502631,
      "warps_active_pct": 14.666908,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.9457971993503962
     }
    }
   ],
   "dominant_kernel": "nvjet_sm90_tst_128x200_64x5_2x1_v_bz_coopA_bias_TNT",
   "primary": "compute_bound",
   "primary_rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
   "alternative_explanation": "Padding/redundant work inside nominal FLOPs; instruction mix vs tensor pipe",
   "counter_experiment": "Same dtype better tile/kernel; compare effective vs executed FLOPs",
   "status": "classified"
  },
  {
   "group": "dit:aten.linear.default:{'M': 1170, 'N': 5120, 'K': 13824}",
   "resolution": "480x832",
   "stage": "dit",
   "op": "aten.linear.default",
   "shape_parameters": {
    "M": 1170,
    "N": 5120,
    "K": 13824
   },
   "input_sha256": "d89811d1f0f1ab9382656380f7ff354f19e521d815cd71b6e39b7d2a72def4ec",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing_ef1fe591ff44862b_v1",
   "scheduler_note": null,
   "nominal_mac_flops": 165622579200,
   "summed_kernel_us": 220.60799999999998,
   "kernels": [
    {
     "kernel": "nvjet_sm90_tst_128x200_64x5_2x1_v_bz_coopA_bias_TNT",
     "metrics": {
      "dur_us": 220.608,
      "sm_pct": 85.960958,
      "tensor_pct": 85.960958,
      "dram_pct": 20.442762,
      "l2_pct": 46.922333,
      "warps_active_pct": 14.415541,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 226.492,
      "smem_static_b": 0.0,
      "l2_hit_pct": 69.925913,
      "sm_cycles_active": 322586.174242,
      "cycles_elapsed": 379932.0,
      "dram_read_mb": 206.323712,
      "dram_write_mb": 10.662912,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 85.960958,
      "tensor_pct": 85.960958,
      "dram_pct": 20.442762,
      "l2_pct": 46.922333,
      "warps_active_pct": 14.415541,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.8490629224229599
     }
    }
   ],
   "dominant_kernel": "nvjet_sm90_tst_128x200_64x5_2x1_v_bz_coopA_bias_TNT",
   "primary": "compute_bound",
   "primary_rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
   "alternative_explanation": "Padding/redundant work inside nominal FLOPs; instruction mix vs tensor pipe",
   "counter_experiment": "Same dtype better tile/kernel; compare effective vs executed FLOPs",
   "status": "classified"
  },
  {
   "group": "dit:aten.linear.default:{'M': 1170, 'N': 5120, 'K': 5120}",
   "resolution": "480x832",
   "stage": "dit",
   "op": "aten.linear.default",
   "shape_parameters": {
    "M": 1170,
    "N": 5120,
    "K": 5120
   },
   "input_sha256": "7de91ad08752fccf4380732c881403df8972bc20e43941ce68d7ee08b8d5d769",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing_6d01e6f8a09a63ac_v1",
   "scheduler_note": null,
   "nominal_mac_flops": 61341696000,
   "summed_kernel_us": 85.696,
   "kernels": [
    {
     "kernel": "nvjet_sm90_tst_128x200_64x5_2x1_v_bz_coopA_bias_TNT",
     "metrics": {
      "dur_us": 85.696,
      "sm_pct": 81.017145,
      "tensor_pct": 81.017145,
      "dram_pct": 20.578043,
      "l2_pct": 44.816086,
      "warps_active_pct": 14.417252,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 226.492,
      "smem_static_b": 0.0,
      "l2_hit_pct": 71.056456,
      "sm_cycles_active": 124644.80303,
      "cycles_elapsed": 149624.0,
      "dram_read_mb": 74.979584,
      "dram_write_mb": 9.85472,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 81.017145,
      "tensor_pct": 81.017145,
      "dram_pct": 20.578043,
      "l2_pct": 44.816086,
      "warps_active_pct": 14.417252,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.8330535410762979
     }
    }
   ],
   "dominant_kernel": "nvjet_sm90_tst_128x200_64x5_2x1_v_bz_coopA_bias_TNT",
   "primary": "compute_bound",
   "primary_rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
   "alternative_explanation": "Padding/redundant work inside nominal FLOPs; instruction mix vs tensor pipe",
   "counter_experiment": "Same dtype better tile/kernel; compare effective vs executed FLOPs",
   "status": "classified"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 192, 6, 242, 106], 'weight': [192, 192, 3, 3, 3], 'output': [1, 192, 4, 240, 104], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "480x832",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     192,
     6,
     242,
     106
    ],
    "weight": [
     192,
     192,
     3,
     3,
     3
    ],
    "output": [
     1,
     192,
     4,
     240,
     104
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "7ea1828a641121c7e50c8fec13aa00d78a4f4b0f419741b8b6e74c3fd8f2b955",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing_41fea95495c28b56_v1",
   "scheduler_note": null,
   "nominal_mac_flops": 198747095040,
   "summed_kernel_us": 872.4159999999999,
   "kernels": [
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 68.352,
      "sm_pct": 39.125069,
      "tensor_pct": 0.0,
      "dram_pct": 66.195507,
      "l2_pct": 82.796731,
      "warps_active_pct": 92.678842,
      "max_warps_pct": 100.0,
      "waves": 27.33,
      "grid": 28860.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 55.889451,
      "sm_cycles_active": 129394.015152,
      "cycles_elapsed": 134903.0,
      "dram_read_mb": 118.218752,
      "dram_write_mb": 99.319552,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "dram_bandwidth_bound",
     "rule": "DRAM throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 39.125069,
      "tensor_pct": 0.0,
      "dram_pct": 66.195507,
      "l2_pct": 82.796731,
      "warps_active_pct": 92.678842,
      "waves": 27.33,
      "sm_active_over_elapsed": 0.959163362949675
     }
    },
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 5.216,
      "sm_pct": 31.247078,
      "tensor_pct": 0.0,
      "dram_pct": 16.025307,
      "l2_pct": 38.455553,
      "warps_active_pct": 74.300598,
      "max_warps_pct": 100.0,
      "waves": 1.09,
      "grid": 1152.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 58.646098,
      "sm_cycles_active": 5233.333333,
      "cycles_elapsed": 10218.0,
      "dram_read_mb": 3.990016,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 31.247078,
      "tensor_pct": 0.0,
      "dram_pct": 16.025307,
      "l2_pct": 38.455553,
      "warps_active_pct": 74.300598,
      "waves": 1.09,
      "sm_active_over_elapsed": 0.5121680693873556
     }
    },
    {
     "kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
     "metrics": {
      "dur_us": 706.24,
      "sm_pct": 78.229995,
      "tensor_pct": 78.229995,
      "dram_pct": 9.312729,
      "l2_pct": 68.377794,
      "warps_active_pct": 18.165615,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 231.424,
      "smem_static_b": 0.0,
      "l2_hit_pct": 101.286436,
      "sm_cycles_active": 1198319.727273,
      "cycles_elapsed": 1261640.0,
      "dram_read_mb": 244.80384,
      "dram_write_mb": 71.66336,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 78.229995,
      "tensor_pct": 78.229995,
      "dram_pct": 9.312729,
      "l2_pct": 68.377794,
      "warps_active_pct": 18.165615,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.9498111404782663
     }
    },
    {
     "kernel": "void cudnn::nhwcToNchwKernel<float, float, float, 1, 0, 0>(cudnn::nhwc2nchw_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 39.008,
      "sm_pct": 40.024653,
      "tensor_pct": 0.0,
      "dram_pct": 70.528428,
      "l2_pct": 85.497306,
      "warps_active_pct": 90.890535,
      "max_warps_pct": 100.0,
      "waves": 17.73,
      "grid": 18720.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 51.039836,
      "sm_cycles_active": 73393.234848,
      "cycles_elapsed": 77111.0,
      "dram_read_mb": 76.694528,
      "dram_write_mb": 55.604736,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "dram_bandwidth_bound",
     "rule": "DRAM throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 40.024653,
      "tensor_pct": 0.0,
      "dram_pct": 70.528428,
      "l2_pct": 85.497306,
      "warps_active_pct": 90.890535,
      "waves": 17.73,
      "sm_active_over_elapsed": 0.9517868377793051
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 53.6,
      "sm_pct": 59.654443,
      "tensor_pct": 2.14699,
      "dram_pct": 49.625097,
      "l2_pct": 62.720143,
      "warps_active_pct": 81.25356,
      "max_warps_pct": 100.0,
      "waves": 35.45,
      "grid": 74880.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 50.551869,
      "sm_cycles_active": 102431.537879,
      "cycles_elapsed": 106019.0,
      "dram_read_mb": 76.694784,
      "dram_write_mb": 51.2256,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "l2_bandwidth_bound",
     "rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
     "evidence": {
      "sm_pct": 59.654443,
      "tensor_pct": 2.14699,
      "dram_pct": 49.625097,
      "l2_pct": 62.720143,
      "warps_active_pct": 81.25356,
      "waves": 35.45,
      "sm_active_over_elapsed": 0.9661620830134221
     }
    }
   ],
   "dominant_kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
   "primary": "compute_bound",
   "primary_rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
   "alternative_explanation": "Padding/redundant work inside nominal FLOPs; instruction mix vs tensor pipe",
   "counter_experiment": "Same dtype better tile/kernel; compare effective vs executed FLOPs",
   "status": "classified"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 96, 6, 482, 210], 'weight': [96, 96, 3, 3, 3], 'output': [1, 96, 4, 480, 208], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "480x832",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     96,
     6,
     482,
     210
    ],
    "weight": [
     96,
     96,
     3,
     3,
     3
    ],
    "output": [
     1,
     96,
     4,
     480,
     208
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "3da5719db48ceca956fb363d6dbfa423964bbde96b9ce8b8dd9399261d9405da",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing_74aff8f080d200c8_v1",
   "scheduler_note": null,
   "nominal_mac_flops": 198747095040,
   "summed_kernel_us": 1672.16,
   "kernels": [
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 0.140096,
      "sm_pct": 38.200711,
      "tensor_pct": 0.0,
      "dram_pct": 66.546526,
      "l2_pct": 81.38646,
      "warps_active_pct": 92.912553,
      "max_warps_pct": 100.0,
      "waves": 53.92,
      "grid": 56937.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 55.373946,
      "sm_cycles_active": 267600.515152,
      "cycles_elapsed": 273173.0,
      "dram_read_mb": 233.226496,
      "dram_write_mb": 215.286784,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "dram_bandwidth_bound",
     "rule": "DRAM throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 38.200711,
      "tensor_pct": 0.0,
      "dram_pct": 66.546526,
      "l2_pct": 81.38646,
      "warps_active_pct": 92.912553,
      "waves": 53.92,
      "sm_active_over_elapsed": 0.9796008944954296
     }
    },
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 0.003648,
      "sm_pct": 11.926683,
      "tensor_pct": 0.0,
      "dram_pct": 5.743217,
      "l2_pct": 14.876259,
      "warps_active_pct": 25.684751,
      "max_warps_pct": 100.0,
      "waves": 0.27,
      "grid": 288.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 58.037528,
      "sm_cycles_active": 3001.015152,
      "cycles_elapsed": 7200.0,
      "dram_read_mb": 1.004288,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 11.926683,
      "tensor_pct": 0.0,
      "dram_pct": 5.743217,
      "l2_pct": 14.876259,
      "warps_active_pct": 25.684751,
      "waves": 0.27,
      "sm_active_over_elapsed": 0.41680765999999997
     }
    },
    {
     "kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize256x32x32_warpgroupsize1x1x1_g1_execute_segment_k",
     "metrics": {
      "dur_us": 1.34,
      "sm_pct": 30.613135,
      "tensor_pct": 30.613135,
      "dram_pct": 9.524368,
      "l2_pct": 59.103268,
      "warps_active_pct": 18.22425,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 231.424,
      "smem_static_b": 0.0,
      "l2_hit_pct": 90.285086,
      "sm_cycles_active": 2379744.893939,
      "cycles_elapsed": 2411969.0,
      "dram_read_mb": 467.56096,
      "dram_write_mb": 146.57536,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 30.613135,
      "tensor_pct": 30.613135,
      "dram_pct": 9.524368,
      "l2_pct": 59.103268,
      "warps_active_pct": 18.22425,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.9866399169885683
     }
    },
    {
     "kernel": "void cudnn::nhwcToNchwKernel<float, float, float, 1, 0, 0>(cudnn::nhwc2nchw_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 0.08096,
      "sm_pct": 41.056031,
      "tensor_pct": 0.0,
      "dram_pct": 73.516066,
      "l2_pct": 84.811005,
      "warps_active_pct": 93.288254,
      "max_warps_pct": 100.0,
      "waves": 35.45,
      "grid": 37440.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 51.307389,
      "sm_cycles_active": 150109.522727,
      "cycles_elapsed": 158588.0,
      "dram_read_mb": 153.371136,
      "dram_write_mb": 132.944384,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "dram_bandwidth_bound",
     "rule": "DRAM throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 41.056031,
      "tensor_pct": 0.0,
      "dram_pct": 73.516066,
      "l2_pct": 84.811005,
      "warps_active_pct": 93.288254,
      "waves": 35.45,
      "sm_active_over_elapsed": 0.9465377123552854
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 0.107456,
      "sm_pct": 60.133901,
      "tensor_pct": 2.165631,
      "dram_pct": 54.32714,
      "l2_pct": 63.797536,
      "warps_active_pct": 83.574473,
      "max_warps_pct": 100.0,
      "waves": 70.91,
      "grid": 149760.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 50.421829,
      "sm_cycles_active": 204876.628788,
      "cycles_elapsed": 210831.0,
      "dram_read_mb": 153.37088,
      "dram_write_mb": 127.412736,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 60.133901,
      "tensor_pct": 2.165631,
      "dram_pct": 54.32714,
      "l2_pct": 63.797536,
      "warps_active_pct": 83.574473,
      "waves": 70.91,
      "sm_active_over_elapsed": 0.9717576105411443
     }
    }
   ],
   "dominant_kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize256x32x32_warpgroupsize1x1x1_g1_execute_segment_k",
   "primary": "unresolved_no_dominant_limiter",
   "primary_rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
   "alternative_explanation": "Mixed limiter or measurement window artifact",
   "counter_experiment": "Collect stall reasons (smsp__warp_issue_stalled_*) and l1tex/lts bytes",
   "status": "unresolved"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 384, 4, 122, 54], 'weight': [384, 384, 3, 3, 3], 'output': [1, 384, 2, 120, 52], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "480x832",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     384,
     4,
     122,
     54
    ],
    "weight": [
     384,
     384,
     3,
     3,
     3
    ],
    "output": [
     1,
     384,
     2,
     120,
     52
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "538cab189b8721f3989e60d331a5e8f3a0b3ea6e3f542b4f5be45a3f1807f370",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing_7fc47a1ee259ec1d_v1",
   "scheduler_note": null,
   "nominal_mac_flops": 99373547520,
   "summed_kernel_us": 486.23999999999995,
   "kernels": [
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 22.304,
      "sm_pct": 41.271243,
      "tensor_pct": 0.0,
      "dram_pct": 56.938789,
      "l2_pct": 79.357066,
      "warps_active_pct": 88.712471,
      "max_warps_pct": 100.0,
      "waves": 9.36,
      "grid": 9888.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 50.871217,
      "sm_cycles_active": 38130.212121,
      "cycles_elapsed": 44026.0,
      "dram_read_mb": 40.485888,
      "dram_write_mb": 20.514304,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "l2_bandwidth_bound",
     "rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
     "evidence": {
      "sm_pct": 41.271243,
      "tensor_pct": 0.0,
      "dram_pct": 56.938789,
      "l2_pct": 79.357066,
      "warps_active_pct": 88.712471,
      "waves": 9.36,
      "sm_active_over_elapsed": 0.8660839531413255
     }
    },
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 10.304,
      "sm_pct": 61.769375,
      "tensor_pct": 0.0,
      "dram_pct": 33.529028,
      "l2_pct": 67.254028,
      "warps_active_pct": 84.865132,
      "max_warps_pct": 100.0,
      "waves": 4.36,
      "grid": 4608.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 56.4358,
      "sm_cycles_active": 15307.310606,
      "cycles_elapsed": 20210.0,
      "dram_read_mb": 15.934464,
      "dram_write_mb": 0.635392,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 61.769375,
      "tensor_pct": 0.0,
      "dram_pct": 33.529028,
      "l2_pct": 67.254028,
      "warps_active_pct": 84.865132,
      "waves": 4.36,
      "sm_active_over_elapsed": 0.7574126969816922
     }
    },
    {
     "kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
     "metrics": {
      "dur_us": 427.488,
      "sm_pct": 65.894637,
      "tensor_pct": 65.894637,
      "dram_pct": 6.41846,
      "l2_pct": 53.429551,
      "warps_active_pct": 14.878182,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 231.424,
      "smem_static_b": 0.0,
      "l2_hit_pct": 89.975772,
      "sm_cycles_active": 668372.340909,
      "cycles_elapsed": 754629.0,
      "dram_read_mb": 113.32224,
      "dram_write_mb": 18.706944,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 65.894637,
      "tensor_pct": 65.894637,
      "dram_pct": 6.41846,
      "l2_pct": 53.429551,
      "warps_active_pct": 14.878182,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.8856966017857782
     }
    },
    {
     "kernel": "void cudnn::nhwcToNchwKernel<float, float, float, 1, 0, 0>(cudnn::nhwc2nchw_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 11.2,
      "sm_pct": 34.827828,
      "tensor_pct": 0.0,
      "dram_pct": 39.394113,
      "l2_pct": 71.450544,
      "warps_active_pct": 86.668335,
      "max_warps_pct": 100.0,
      "waves": 4.43,
      "grid": 4680.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 51.780723,
      "sm_cycles_active": 17123.25,
      "cycles_elapsed": 22041.0,
      "dram_read_mb": 19.185664,
      "dram_write_mb": 1.973248,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "l2_bandwidth_bound",
     "rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
     "evidence": {
      "sm_pct": 34.827828,
      "tensor_pct": 0.0,
      "dram_pct": 39.394113,
      "l2_pct": 71.450544,
      "warps_active_pct": 86.668335,
      "waves": 4.43,
      "sm_active_over_elapsed": 0.7768817204301075
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 14.944,
      "sm_pct": 54.104812,
      "tensor_pct": 1.940509,
      "dram_pct": 26.885869,
      "l2_pct": 57.817172,
      "warps_active_pct": 75.631676,
      "max_warps_pct": 100.0,
      "waves": 8.86,
      "grid": 18720.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 51.432523,
      "sm_cycles_active": 25121.856061,
      "cycles_elapsed": 29455.0,
      "dram_read_mb": 19.185408,
      "dram_write_mb": 0.086016,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 54.104812,
      "tensor_pct": 1.940509,
      "dram_pct": 26.885869,
      "l2_pct": 57.817172,
      "warps_active_pct": 75.631676,
      "waves": 8.86,
      "sm_active_over_elapsed": 0.8528893587166865
     }
    }
   ],
   "dominant_kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
   "primary": "compute_bound",
   "primary_rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
   "alternative_explanation": "Padding/redundant work inside nominal FLOPs; instruction mix vs tensor pipe",
   "counter_experiment": "Same dtype better tile/kernel; compare effective vs executed FLOPs",
   "status": "classified"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 192, 4, 122, 54], 'weight': [384, 192, 3, 3, 3], 'output': [1, 384, 2, 120, 52], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "480x832",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     192,
     4,
     122,
     54
    ],
    "weight": [
     384,
     192,
     3,
     3,
     3
    ],
    "output": [
     1,
     384,
     2,
     120,
     52
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "6fa08b3e20b8b3a0f9f34750ff5afbc4a19f18369bd70ccbf0ed0d1071ca5f08",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing_e49e1cfe69d26b07_v1",
   "scheduler_note": null,
   "nominal_mac_flops": 49686773760,
   "summed_kernel_us": 253.02399999999997,
   "kernels": [
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 11.456,
      "sm_pct": 40.528654,
      "tensor_pct": 0.0,
      "dram_pct": 42.349603,
      "l2_pct": 75.323444,
      "warps_active_pct": 85.832553,
      "max_warps_pct": 100.0,
      "waves": 4.68,
      "grid": 4944.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 51.628655,
      "sm_cycles_active": 18604.69697,
      "cycles_elapsed": 22559.0,
      "dram_read_mb": 20.249088,
      "dram_write_mb": 3.02208,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "l2_bandwidth_bound",
     "rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
     "evidence": {
      "sm_pct": 40.528654,
      "tensor_pct": 0.0,
      "dram_pct": 42.349603,
      "l2_pct": 75.323444,
      "warps_active_pct": 85.832553,
      "waves": 4.68,
      "sm_active_over_elapsed": 0.8247128405514429
     }
    },
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 6.656,
      "sm_pct": 48.175667,
      "tensor_pct": 0.0,
      "dram_pct": 25.012249,
      "l2_pct": 54.92529,
      "warps_active_pct": 81.48079,
      "max_warps_pct": 100.0,
      "waves": 2.18,
      "grid": 2304.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 56.700541,
      "sm_cycles_active": 8695.643939,
      "cycles_elapsed": 13055.0,
      "dram_read_mb": 7.971328,
      "dram_write_mb": 0.000768,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 48.175667,
      "tensor_pct": 0.0,
      "dram_pct": 25.012249,
      "l2_pct": 54.92529,
      "warps_active_pct": 81.48079,
      "waves": 2.18,
      "sm_active_over_elapsed": 0.6660776667177326
     }
    },
    {
     "kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
     "metrics": {
      "dur_us": 209.216,
      "sm_pct": 68.580954,
      "tensor_pct": 68.580954,
      "dram_pct": 5.12333,
      "l2_pct": 52.419661,
      "warps_active_pct": 14.794025,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 231.424,
      "smem_static_b": 0.0,
      "l2_hit_pct": 92.54203,
      "sm_cycles_active": 328062.371212,
      "cycles_elapsed": 362840.0,
      "dram_read_mb": 38.672128,
      "dram_write_mb": 12.901632,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 68.580954,
      "tensor_pct": 68.580954,
      "dram_pct": 5.12333,
      "l2_pct": 52.419661,
      "warps_active_pct": 14.794025,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.9041516128651748
     }
    },
    {
     "kernel": "void cudnn::nhwcToNchwKernel<float, float, float, 1, 0, 0>(cudnn::nhwc2nchw_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 11.136,
      "sm_pct": 35.060846,
      "tensor_pct": 0.0,
      "dram_pct": 40.414344,
      "l2_pct": 72.631446,
      "warps_active_pct": 85.495716,
      "max_warps_pct": 100.0,
      "waves": 4.43,
      "grid": 4680.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 51.789359,
      "sm_cycles_active": 17216.80303,
      "cycles_elapsed": 21905.0,
      "dram_read_mb": 19.185664,
      "dram_write_mb": 2.391552,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "l2_bandwidth_bound",
     "rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
     "evidence": {
      "sm_pct": 35.060846,
      "tensor_pct": 0.0,
      "dram_pct": 40.414344,
      "l2_pct": 72.631446,
      "warps_active_pct": 85.495716,
      "waves": 4.43,
      "sm_active_over_elapsed": 0.7859759429354028
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 14.56,
      "sm_pct": 55.602228,
      "tensor_pct": 1.993626,
      "dram_pct": 27.777187,
      "l2_pct": 59.956106,
      "warps_active_pct": 77.608876,
      "max_warps_pct": 100.0,
      "waves": 8.86,
      "grid": 18720.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 51.849463,
      "sm_cycles_active": 24635.219697,
      "cycles_elapsed": 28689.0,
      "dram_read_mb": 19.185152,
      "dram_write_mb": 0.22272,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 55.602228,
      "tensor_pct": 1.993626,
      "dram_pct": 27.777187,
      "l2_pct": 59.956106,
      "warps_active_pct": 77.608876,
      "waves": 8.86,
      "sm_active_over_elapsed": 0.8586991424239255
     }
    }
   ],
   "dominant_kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
   "primary": "compute_bound",
   "primary_rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
   "alternative_explanation": "Padding/redundant work inside nominal FLOPs; instruction mix vs tensor pipe",
   "counter_experiment": "Same dtype better tile/kernel; compare effective vs executed FLOPs",
   "status": "classified"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 384, 4, 120, 52], 'weight': [768, 384, 3, 1, 1], 'output': [1, 768, 2, 120, 52], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "480x832",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     384,
     4,
     120,
     52
    ],
    "weight": [
     768,
     384,
     3,
     1,
     1
    ],
    "output": [
     1,
     768,
     2,
     120,
     52
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "3547fbf5eb18cf7da7abff35f3a785e4dd1cfc51a243c23698a162ff25086677",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing_c87dc674a51efcf2_v1",
   "scheduler_note": null,
   "nominal_mac_flops": 22083010560,
   "summed_kernel_us": 168.224,
   "kernels": [
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 20.224,
      "sm_pct": 43.18329,
      "tensor_pct": 0.0,
      "dram_pct": 58.463134,
      "l2_pct": 81.060841,
      "warps_active_pct": 91.642053,
      "max_warps_pct": 100.0,
      "waves": 8.86,
      "grid": 9360.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 50.924177,
      "sm_cycles_active": 35113.628788,
      "cycles_elapsed": 39886.0,
      "dram_read_mb": 38.349568,
      "dram_write_mb": 18.360832,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "l2_bandwidth_bound",
     "rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
     "evidence": {
      "sm_pct": 43.18329,
      "tensor_pct": 0.0,
      "dram_pct": 58.463134,
      "l2_pct": 81.060841,
      "warps_active_pct": 91.642053,
      "waves": 8.86,
      "sm_active_over_elapsed": 0.8803497163917164
     }
    },
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 15.2,
      "sm_pct": 82.620515,
      "tensor_pct": 0.0,
      "dram_pct": 4.858398,
      "l2_pct": 12.826662,
      "warps_active_pct": 86.569261,
      "max_warps_pct": 100.0,
      "waves": 8.73,
      "grid": 9216.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 66.646764,
      "sm_cycles_active": 25683.340909,
      "cycles_elapsed": 30040.0,
      "dram_read_mb": 3.547904,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 82.620515,
      "tensor_pct": 0.0,
      "dram_pct": 4.858398,
      "l2_pct": 12.826662,
      "warps_active_pct": 86.569261,
      "waves": 8.73,
      "sm_active_over_elapsed": 0.8549714017643142
     }
    },
    {
     "kernel": "void cask_plugin__5x_cudnn::init_device_workspace_kernel<xmma__5x_cudnn::Warp_specialized_params_non_template<xmma__5x_c",
     "metrics": {
      "dur_us": 3.52,
      "sm_pct": 0.027775,
      "tensor_pct": 0.0,
      "dram_pct": 0.02746,
      "l2_pct": 0.311923,
      "warps_active_pct": 1.554039,
      "max_warps_pct": 50.0,
      "waves": 0.0,
      "grid": 1.0,
      "block": 1.0,
      "regs": 16.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 46.597859,
      "sm_cycles_active": 25.181818,
      "cycles_elapsed": 6907.0,
      "dram_read_mb": 0.004608,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 128.0,
      "occ_limit_smem": 32.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 0.027775,
      "tensor_pct": 0.0,
      "dram_pct": 0.02746,
      "l2_pct": 0.311923,
      "warps_active_pct": 1.554039,
      "waves": 0.0,
      "sm_active_over_elapsed": 0.003645840162154336
     }
    },
    {
     "kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
     "metrics": {
      "dur_us": 81.536,
      "sm_pct": 60.214853,
      "tensor_pct": 60.214853,
      "dram_pct": 23.653585,
      "l2_pct": 58.177611,
      "warps_active_pct": 16.859649,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 231.424,
      "smem_static_b": 0.0,
      "l2_hit_pct": 82.85676,
      "sm_cycles_active": 106282.007576,
      "cycles_elapsed": 138349.0,
      "dram_read_mb": 59.979008,
      "dram_write_mb": 32.811008,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 60.214853,
      "tensor_pct": 60.214853,
      "dram_pct": 23.653585,
      "l2_pct": 58.177611,
      "warps_active_pct": 16.859649,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.7682166663727241
     }
    },
    {
     "kernel": "void cudnn::nhwcToNchwKernel<float, float, float, 1, 0, 0>(cudnn::nhwc2nchw_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 20.032,
      "sm_pct": 38.181248,
      "tensor_pct": 0.0,
      "dram_pct": 57.907027,
      "l2_pct": 81.941375,
      "warps_active_pct": 88.130031,
      "max_warps_pct": 100.0,
      "waves": 8.86,
      "grid": 9360.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 51.046134,
      "sm_cycles_active": 34472.984848,
      "cycles_elapsed": 39558.0,
      "dram_read_mb": 38.358784,
      "dram_write_mb": 17.398784,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "l2_bandwidth_bound",
     "rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
     "evidence": {
      "sm_pct": 38.181248,
      "tensor_pct": 0.0,
      "dram_pct": 57.907027,
      "l2_pct": 81.941375,
      "warps_active_pct": 88.130031,
      "waves": 8.86,
      "sm_active_over_elapsed": 0.8714541899994944
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 27.712,
      "sm_pct": 57.998715,
      "tensor_pct": 2.084501,
      "dram_pct": 38.31248,
      "l2_pct": 59.274381,
      "warps_active_pct": 80.19311,
      "max_warps_pct": 100.0,
      "waves": 17.73,
      "grid": 37440.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 50.860165,
      "sm_cycles_active": 50510.045455,
      "cycles_elapsed": 54736.0,
      "dram_read_mb": 38.356224,
      "dram_write_mb": 12.673792,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 57.998715,
      "tensor_pct": 2.084501,
      "dram_pct": 38.31248,
      "l2_pct": 59.274381,
      "warps_active_pct": 80.19311,
      "waves": 17.73,
      "sm_active_over_elapsed": 0.9227938734105524
     }
    }
   ],
   "dominant_kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
   "primary": "compute_bound",
   "primary_rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
   "alternative_explanation": "Padding/redundant work inside nominal FLOPs; instruction mix vs tensor pipe",
   "counter_experiment": "Same dtype better tile/kernel; compare effective vs executed FLOPs",
   "status": "classified"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 96, 6, 482, 210], 'weight': [3, 96, 3, 3, 3], 'output': [1, 3, 4, 480, 208], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "480x832",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     96,
     6,
     482,
     210
    ],
    "weight": [
     3,
     96,
     3,
     3,
     3
    ],
    "output": [
     1,
     3,
     4,
     480,
     208
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "3558a52a268a76189298b60a739c4bc3a01099db809b69c10633a6c964ffd111",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing_27b8a5265a151590_v1",
   "scheduler_note": null,
   "nominal_mac_flops": 6210846720,
   "summed_kernel_us": 704.128,
   "kernels": [
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 2>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 139.52,
      "sm_pct": 38.236556,
      "tensor_pct": 0.636739,
      "dram_pct": 66.678149,
      "l2_pct": 82.169336,
      "warps_active_pct": 93.600613,
      "max_warps_pct": 100.0,
      "waves": 53.92,
      "grid": 56937.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 55.444303,
      "sm_cycles_active": 265726.848485,
      "cycles_elapsed": 272731.0,
      "dram_read_mb": 233.225728,
      "dram_write_mb": 214.398976,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "dram_bandwidth_bound",
     "rule": "DRAM throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 38.236556,
      "tensor_pct": 0.636739,
      "dram_pct": 66.678149,
      "l2_pct": 82.169336,
      "warps_active_pct": 93.600613,
      "waves": 53.92,
      "sm_active_over_elapsed": 0.9743184620926848
     }
    },
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 2>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 3.2,
      "sm_pct": 0.366427,
      "tensor_pct": 0.004346,
      "dram_pct": 0.266516,
      "l2_pct": 0.908733,
      "warps_active_pct": 12.423183,
      "max_warps_pct": 100.0,
      "waves": 0.01,
      "grid": 9.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 44.489458,
      "sm_cycles_active": 186.454545,
      "cycles_elapsed": 6281.0,
      "dram_read_mb": 0.040704,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 0.366427,
      "tensor_pct": 0.004346,
      "dram_pct": 0.266516,
      "l2_pct": 0.908733,
      "warps_active_pct": 12.423183,
      "waves": 0.01,
      "sm_active_over_elapsed": 0.029685487183569496
     }
    },
    {
     "kernel": "sm80_xmma_fprop_implicit_gemm_indexed_wo_smem_tf32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x16x64_stage1_warpsize4x1x1_g",
     "metrics": {
      "dur_us": 536.64,
      "sm_pct": 49.269089,
      "tensor_pct": 19.44649,
      "dram_pct": 20.583373,
      "l2_pct": 69.006466,
      "warps_active_pct": 29.756278,
      "max_warps_pct": 31.25,
      "waves": 4.73,
      "grid": 3120.0,
      "block": 128.0,
      "regs": 94.0,
      "smem_dyn_kb": 4.096,
      "smem_static_b": 0.0,
      "l2_hit_pct": 74.622305,
      "sm_cycles_active": 911274.219697,
      "cycles_elapsed": 961852.0,
      "dram_read_mb": 523.764736,
      "dram_write_mb": 7.751424,
      "occ_limit_regs": 5.0,
      "occ_limit_smem": 12.0
     },
     "classification": "l2_bandwidth_bound",
     "rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
     "evidence": {
      "sm_pct": 49.269089,
      "tensor_pct": 19.44649,
      "dram_pct": 20.583373,
      "l2_pct": 69.006466,
      "warps_active_pct": 29.756278,
      "waves": 4.73,
      "sm_active_over_elapsed": 0.9474162549924521
     }
    },
    {
     "kernel": "void cudnn::nhwcToNchwKernel<float, float, float, 1, 0, 0>(cudnn::nhwc2nchw_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 18.784,
      "sm_pct": 74.116556,
      "tensor_pct": 0.0,
      "dram_pct": 5.334298,
      "l2_pct": 13.932529,
      "warps_active_pct": 86.846153,
      "max_warps_pct": 100.0,
      "waves": 11.82,
      "grid": 12480.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 66.094296,
      "sm_cycles_active": 32237.5,
      "cycles_elapsed": 37058.0,
      "dram_read_mb": 4.807424,
      "dram_write_mb": 0.000256,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 74.116556,
      "tensor_pct": 0.0,
      "dram_pct": 5.334298,
      "l2_pct": 13.932529,
      "warps_active_pct": 86.846153,
      "waves": 11.82,
      "sm_active_over_elapsed": 0.8699201252091316
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 5.984,
      "sm_pct": 34.462223,
      "tensor_pct": 1.216469,
      "dram_pct": 16.846614,
      "l2_pct": 38.220111,
      "warps_active_pct": 68.534774,
      "max_warps_pct": 100.0,
      "waves": 2.22,
      "grid": 4680.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 52.212473,
      "sm_cycles_active": 8045.401515,
      "cycles_elapsed": 11718.0,
      "dram_read_mb": 4.806656,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 34.462223,
      "tensor_pct": 1.216469,
      "dram_pct": 16.846614,
      "l2_pct": 38.220111,
      "warps_active_pct": 68.534774,
      "waves": 2.22,
      "sm_active_over_elapsed": 0.6865848707117255
     }
    }
   ],
   "dominant_kernel": "sm80_xmma_fprop_implicit_gemm_indexed_wo_smem_tf32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x16x64_stage1_warpsize4x1x1_g",
   "primary": "l2_bandwidth_bound",
   "primary_rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
   "alternative_explanation": "Working set thrashing L2; DRAM counter under-reads write-back",
   "counter_experiment": "Blocking/reuse change; compare lts bytes",
   "status": "classified"
  },
  {
   "group": "dit:aten.linear.default:{'M': 4680, 'N': 64, 'K': 5120}",
   "resolution": "480x832",
   "stage": "dit",
   "op": "aten.linear.default",
   "shape_parameters": {
    "M": 4680,
    "N": 64,
    "K": 5120
   },
   "input_sha256": "34457f152c46e24a2e317b09c8ddc97428430b2e02f99cc2ddeb74c8c157ed9e",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing_301783745952cfe0_v1",
   "scheduler_note": null,
   "nominal_mac_flops": 3067084800,
   "summed_kernel_us": 17.6,
   "kernels": [
    {
     "kernel": "nvjet_sm90_tst_64x48_64x15_1x4_h_bz_bias_TNT",
     "metrics": {
      "dur_us": 17.6,
      "sm_pct": 18.383383,
      "tensor_pct": 18.383383,
      "dram_pct": 60.982457,
      "l2_pct": 62.360801,
      "warps_active_pct": 13.831379,
      "max_warps_pct": 18.75,
      "waves": 0.76,
      "grid": 100.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 231.868,
      "smem_static_b": 0.0,
      "l2_hit_pct": 19.546279,
      "sm_cycles_active": 18972.659091,
      "cycles_elapsed": 31965.0,
      "dram_read_mb": 48.598016,
      "dram_write_mb": 3.023104,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "dram_bandwidth_bound",
     "rule": "DRAM throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 18.383383,
      "tensor_pct": 18.383383,
      "dram_pct": 60.982457,
      "l2_pct": 62.360801,
      "warps_active_pct": 13.831379,
      "waves": 0.76,
      "sm_active_over_elapsed": 0.5935447862036602
     }
    }
   ],
   "dominant_kernel": "nvjet_sm90_tst_64x48_64x15_1x4_h_bz_bias_TNT",
   "primary": "dram_bandwidth_bound",
   "primary_rule": "DRAM throughput >= 60% of peak sustained",
   "alternative_explanation": "Cache-unfriendly layout or spills inflating traffic",
   "counter_experiment": "Layout/fusion that lowers DRAM bytes; check bytes and time fall together",
   "status": "classified"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 384, 3, 60, 26], 'weight': [768, 384, 3, 1, 1], 'output': [1, 768, 1, 60, 26], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "480x832",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     384,
     3,
     60,
     26
    ],
    "weight": [
     768,
     384,
     3,
     1,
     1
    ],
    "output": [
     1,
     768,
     1,
     60,
     26
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "03221b37d95f7c65a91055eae771f796fbf8e0d7758ce6840701ea0c066358ea",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing_d35bb120bedeb7d0_v1",
   "scheduler_note": null,
   "nominal_mac_flops": 2760376320,
   "summed_kernel_us": 51.711999999999996,
   "kernels": [
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 6.144,
      "sm_pct": 27.570005,
      "tensor_pct": 0.0,
      "dram_pct": 24.455907,
      "l2_pct": 53.828489,
      "warps_active_pct": 68.387697,
      "max_warps_pct": 100.0,
      "waves": 1.67,
      "grid": 1764.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 56.95306,
      "sm_cycles_active": 8272.840909,
      "cycles_elapsed": 12098.0,
      "dram_read_mb": 7.19744,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 27.570005,
      "tensor_pct": 0.0,
      "dram_pct": 24.455907,
      "l2_pct": 53.828489,
      "warps_active_pct": 68.387697,
      "waves": 1.67,
      "sm_active_over_elapsed": 0.6838188881633328
     }
    },
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 15.328,
      "sm_pct": 82.067119,
      "tensor_pct": 0.0,
      "dram_pct": 4.830805,
      "l2_pct": 13.075698,
      "warps_active_pct": 84.833643,
      "max_warps_pct": 100.0,
      "waves": 8.73,
      "grid": 9216.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 66.314179,
      "sm_cycles_active": 25815.590909,
      "cycles_elapsed": 30222.0,
      "dram_read_mb": 3.547904,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 82.067119,
      "tensor_pct": 0.0,
      "dram_pct": 4.830805,
      "l2_pct": 13.075698,
      "warps_active_pct": 84.833643,
      "waves": 8.73,
      "sm_active_over_elapsed": 0.8541986271259348
     }
    },
    {
     "kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
     "metrics": {
      "dur_us": 18.688,
      "sm_pct": 32.031099,
      "tensor_pct": 32.031099,
      "dram_pct": 12.041761,
      "l2_pct": 31.760624,
      "warps_active_pct": 11.61247,
      "max_warps_pct": 18.75,
      "waves": 0.59,
      "grid": 78.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 231.424,
      "smem_static_b": 0.0,
      "l2_hit_pct": 73.443041,
      "sm_cycles_active": 17451.590909,
      "cycles_elapsed": 34267.0,
      "dram_read_mb": 10.793216,
      "dram_write_mb": 1.792,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 32.031099,
      "tensor_pct": 32.031099,
      "dram_pct": 12.041761,
      "l2_pct": 31.760624,
      "warps_active_pct": 11.61247,
      "waves": 0.59,
      "sm_active_over_elapsed": 0.5092827183295882
     }
    },
    {
     "kernel": "void cudnn::nhwcToNchwKernel<float, float, float, 1, 0, 0>(cudnn::nhwc2nchw_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 5.376,
      "sm_pct": 18.836147,
      "tensor_pct": 0.0,
      "dram_pct": 18.627188,
      "l2_pct": 42.822127,
      "warps_active_pct": 75.330003,
      "max_warps_pct": 100.0,
      "waves": 1.11,
      "grid": 1176.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 53.180024,
      "sm_cycles_active": 5634.840909,
      "cycles_elapsed": 10610.0,
      "dram_read_mb": 4.807424,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 18.836147,
      "tensor_pct": 0.0,
      "dram_pct": 18.627188,
      "l2_pct": 42.822127,
      "warps_active_pct": 75.330003,
      "waves": 1.11,
      "sm_active_over_elapsed": 0.5310877388312912
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 6.176,
      "sm_pct": 33.110637,
      "tensor_pct": 1.169497,
      "dram_pct": 16.232188,
      "l2_pct": 37.095517,
      "warps_active_pct": 69.871775,
      "max_warps_pct": 100.0,
      "waves": 2.22,
      "grid": 4680.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 52.642603,
      "sm_cycles_active": 8197.613636,
      "cycles_elapsed": 12187.0,
      "dram_read_mb": 4.809728,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 33.110637,
      "tensor_pct": 1.169497,
      "dram_pct": 16.232188,
      "l2_pct": 37.095517,
      "warps_active_pct": 69.871775,
      "waves": 2.22,
      "sm_active_over_elapsed": 0.6726523045868549
     }
    }
   ],
   "dominant_kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
   "primary": "partial_wave_tail",
   "primary_rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
   "alternative_explanation": "Small M for this rank shard; multi-kernel split",
   "counter_experiment": "Batch/tile change or larger per-rank M; compare waves and duration",
   "status": "classified"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 192, 2, 120, 52], 'weight': [384, 192, 1, 1, 1], 'output': [1, 384, 2, 120, 52], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "480x832",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     192,
     2,
     120,
     52
    ],
    "weight": [
     384,
     192,
     1,
     1,
     1
    ],
    "output": [
     1,
     384,
     2,
     120,
     52
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "5da1ee3cce399169645cbf03383f0c7594f9d9b878d0bc38e2e663bb1e06edd8",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing_6a92f0b3792a465a_v1",
   "scheduler_note": null,
   "nominal_mac_flops": 1840250880,
   "summed_kernel_us": 43.936,
   "kernels": [
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::direct_copy_kernel_cuda(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 8.96,
      "sm_pct": 46.570666,
      "tensor_pct": 3.247528,
      "dram_pct": 22.502958,
      "l2_pct": 50.005082,
      "warps_active_pct": 75.221106,
      "max_warps_pct": 100.0,
      "waves": 4.43,
      "grid": 9360.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 52.518655,
      "sm_cycles_active": 13856.742424,
      "cycles_elapsed": 17589.0,
      "dram_read_mb": 9.59744,
      "dram_write_mb": 0.043264,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 46.570666,
      "tensor_pct": 3.247528,
      "dram_pct": 22.502958,
      "l2_pct": 50.005082,
      "warps_active_pct": 75.221106,
      "waves": 4.43,
      "sm_active_over_elapsed": 0.7878072900108022
     }
    },
    {
     "kernel": "void cutlass::Kernel2<cutlass_80_tensorop_s1688gemm_128x64_16x6_nn_align4>(Params)",
     "metrics": {
      "dur_us": 20.0,
      "sm_pct": 35.772448,
      "tensor_pct": 28.805683,
      "dram_pct": 12.288959,
      "l2_pct": 42.509332,
      "warps_active_pct": 16.049719,
      "max_warps_pct": 18.75,
      "waves": 1.58,
      "grid": 624.0,
      "block": 128.0,
      "regs": 136.0,
      "smem_dyn_kb": 73.728,
      "smem_static_b": 0.0,
      "l2_hit_pct": 82.511864,
      "sm_cycles_active": 29373.174242,
      "cycles_elapsed": 35839.0,
      "dram_read_mb": 9.906432,
      "dram_write_mb": 1.899264,
      "occ_limit_regs": 3.0,
      "occ_limit_smem": 3.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 35.772448,
      "tensor_pct": 28.805683,
      "dram_pct": 12.288959,
      "l2_pct": 42.509332,
      "warps_active_pct": 16.049719,
      "waves": 1.58,
      "sm_active_over_elapsed": 0.8195868813861994
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 14.976,
      "sm_pct": 53.944847,
      "tensor_pct": 1.934023,
      "dram_pct": 27.158949,
      "l2_pct": 57.739165,
      "warps_active_pct": 76.844231,
      "max_warps_pct": 100.0,
      "waves": 8.86,
      "grid": 18720.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 52.039314,
      "sm_cycles_active": 24719.719697,
      "cycles_elapsed": 29566.0,
      "dram_read_mb": 19.185408,
      "dram_write_mb": 0.368896,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 53.944847,
      "tensor_pct": 1.934023,
      "dram_pct": 27.158949,
      "l2_pct": 57.739165,
      "warps_active_pct": 76.844231,
      "waves": 8.86,
      "sm_active_over_elapsed": 0.8360860345329094
     }
    }
   ],
   "dominant_kernel": "void cutlass::Kernel2<cutlass_80_tensorop_s1688gemm_128x64_16x6_nn_align4>(Params)",
   "primary": "unresolved_no_dominant_limiter",
   "primary_rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
   "alternative_explanation": "Mixed limiter or measurement window artifact",
   "counter_experiment": "Collect stall reasons (smsp__warp_issue_stalled_*) and l1tex/lts bytes",
   "status": "unresolved"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 96, 3, 482, 210], 'weight': [3, 96, 3, 3, 3], 'output': [1, 3, 1, 480, 208], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "480x832",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     96,
     3,
     482,
     210
    ],
    "weight": [
     3,
     96,
     3,
     3,
     3
    ],
    "output": [
     1,
     3,
     1,
     480,
     208
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "480f421d034402c9b8b81d025dce79d2061ec150ba4509115415f8dd5d3b37b5",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing_ddb198f9c3730e1d_v1",
   "scheduler_note": null,
   "nominal_mac_flops": 1552711680,
   "summed_kernel_us": 284.0,
   "kernels": [
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 2>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 70.304,
      "sm_pct": 37.491615,
      "tensor_pct": 0.623769,
      "dram_pct": 63.85219,
      "l2_pct": 80.79348,
      "warps_active_pct": 91.921425,
      "max_warps_pct": 100.0,
      "waves": 26.96,
      "grid": 28470.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 57.456713,
      "sm_cycles_active": 133848.227273,
      "cycles_elapsed": 138850.0,
      "dram_read_mb": 116.620288,
      "dram_write_mb": 99.355648,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "dram_bandwidth_bound",
     "rule": "DRAM throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 37.491615,
      "tensor_pct": 0.623769,
      "dram_pct": 63.85219,
      "l2_pct": 80.79348,
      "warps_active_pct": 91.921425,
      "waves": 26.96,
      "sm_active_over_elapsed": 0.9639771499675909
     }
    },
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 2>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 3.2,
      "sm_pct": 0.368443,
      "tensor_pct": 0.00437,
      "dram_pct": 0.266028,
      "l2_pct": 1.098645,
      "warps_active_pct": 11.62121,
      "max_warps_pct": 100.0,
      "waves": 0.01,
      "grid": 9.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 52.29458,
      "sm_cycles_active": 190.651515,
      "cycles_elapsed": 6247.0,
      "dram_read_mb": 0.040448,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 0.368443,
      "tensor_pct": 0.00437,
      "dram_pct": 0.266028,
      "l2_pct": 1.098645,
      "warps_active_pct": 11.62121,
      "waves": 0.01,
      "sm_active_over_elapsed": 0.030518891467904593
     }
    },
    {
     "kernel": "sm80_xmma_fprop_implicit_gemm_indexed_wo_smem_tf32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x16x64_stage1_warpsize4x1x1_g",
     "metrics": {
      "dur_us": 200.416,
      "sm_pct": 32.875367,
      "tensor_pct": 12.967633,
      "dram_pct": 12.640777,
      "l2_pct": 41.635844,
      "warps_active_pct": 24.081679,
      "max_warps_pct": 31.25,
      "waves": 1.18,
      "grid": 780.0,
      "block": 128.0,
      "regs": 94.0,
      "smem_dyn_kb": 4.096,
      "smem_static_b": 0.0,
      "l2_hit_pct": 77.668223,
      "sm_cycles_active": 284389.871212,
      "cycles_elapsed": 361795.0,
      "dram_read_mb": 117.859072,
      "dram_write_mb": 4.027904,
      "occ_limit_regs": 5.0,
      "occ_limit_smem": 12.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 32.875367,
      "tensor_pct": 12.967633,
      "dram_pct": 12.640777,
      "l2_pct": 41.635844,
      "warps_active_pct": 24.081679,
      "waves": 1.18,
      "sm_active_over_elapsed": 0.7860525192774914
     }
    },
    {
     "kernel": "void cudnn::nhwcToNchwKernel<float, float, float, 1, 0, 0>(cudnn::nhwc2nchw_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 6.496,
      "sm_pct": 53.903145,
      "tensor_pct": 0.0,
      "dram_pct": 3.897877,
      "l2_pct": 11.51182,
      "warps_active_pct": 81.875792,
      "max_warps_pct": 100.0,
      "waves": 2.95,
      "grid": 3120.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 68.220553,
      "sm_cycles_active": 9115.454545,
      "cycles_elapsed": 12798.0,
      "dram_read_mb": 1.21344,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 53.903145,
      "tensor_pct": 0.0,
      "dram_pct": 3.897877,
      "l2_pct": 11.51182,
      "warps_active_pct": 81.875792,
      "waves": 2.95,
      "sm_active_over_elapsed": 0.7122561763556806
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 3.584,
      "sm_pct": 15.263738,
      "tensor_pct": 0.506388,
      "dram_pct": 7.117743,
      "l2_pct": 16.514775,
      "warps_active_pct": 50.054617,
      "max_warps_pct": 100.0,
      "waves": 0.55,
      "grid": 1170.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 52.891471,
      "sm_cycles_active": 3509.257576,
      "cycles_elapsed": 7024.0,
      "dram_read_mb": 1.212672,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 15.263738,
      "tensor_pct": 0.506388,
      "dram_pct": 7.117743,
      "l2_pct": 16.514775,
      "warps_active_pct": 50.054617,
      "waves": 0.55,
      "sm_active_over_elapsed": 0.49960956378132115
     }
    }
   ],
   "dominant_kernel": "sm80_xmma_fprop_implicit_gemm_indexed_wo_smem_tf32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x16x64_stage1_warpsize4x1x1_g",
   "primary": "partial_wave_tail",
   "primary_rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
   "alternative_explanation": "Small M for this rank shard; multi-kernel split",
   "counter_experiment": "Batch/tile change or larger per-rank M; compare waves and duration",
   "status": "classified"
  },
  {
   "group": "dit:aten.linear.default:{'M': 2640, 'N': 5120, 'K': 5120}",
   "resolution": "704x1280",
   "stage": "dit",
   "op": "aten.linear.default",
   "shape_parameters": {
    "M": 2640,
    "N": 5120,
    "K": 5120
   },
   "input_sha256": "8e60010c4ab22e3ad90412ff00e301e28227cf54db05320e456d91009e843787",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing704_255d8b6427b0b5a7_v1",
   "scheduler_note": [],
   "nominal_mac_flops": 138412032000,
   "summed_kernel_us": 185.344,
   "kernels": [
    {
     "kernel": "nvjet_sm90_tst_192x192_64x4_1x2_h_bz_coopB_bias_TNN",
     "metrics": {
      "dur_us": 185.344,
      "sm_pct": 87.977072,
      "tensor_pct": 87.977072,
      "dram_pct": 20.855918,
      "l2_pct": 42.173961,
      "warps_active_pct": 14.524666,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 213.204,
      "smem_static_b": 0.0,
      "l2_hit_pct": 73.305503,
      "sm_cycles_active": 279664.833333,
      "cycles_elapsed": 303601.0,
      "dram_read_mb": 161.596672,
      "dram_write_mb": 24.357888,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 87.977072,
      "tensor_pct": 87.977072,
      "dram_pct": 20.855918,
      "l2_pct": 42.173961,
      "warps_active_pct": 14.524666,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.9211591310074737
     }
    }
   ],
   "dominant_kernel": "nvjet_sm90_tst_192x192_64x4_1x2_h_bz_coopB_bias_TNN",
   "primary": "compute_bound",
   "primary_rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
   "alternative_explanation": "Padding/redundant work inside nominal FLOPs; instruction mix vs tensor pipe",
   "counter_experiment": "Same dtype better tile/kernel; compare effective vs executed FLOPs",
   "status": "classified"
  },
  {
   "group": "dit:aten.linear.default:{'M': 2640, 'N': 15360, 'K': 5120}",
   "resolution": "704x1280",
   "stage": "dit",
   "op": "aten.linear.default",
   "shape_parameters": {
    "M": 2640,
    "N": 15360,
    "K": 5120
   },
   "input_sha256": "5486c3201c4910ee4de3cbdc2b72d468ede3117a03f1e731537cf900e60ecc9d",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing704_db47f7eb1f64a850_v1",
   "scheduler_note": [],
   "nominal_mac_flops": 415236096000,
   "summed_kernel_us": 518.112,
   "kernels": [
    {
     "kernel": "nvjet_sm90_tst_192x208_64x4_2x1_v_bz_coopB_bias_TNT",
     "metrics": {
      "dur_us": 518.112,
      "sm_pct": 92.989233,
      "tensor_pct": 92.989233,
      "dram_pct": 21.168798,
      "l2_pct": 47.743474,
      "warps_active_pct": 14.724952,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 221.396,
      "smem_static_b": 0.0,
      "l2_hit_pct": 73.456413,
      "sm_cycles_active": 820164.666667,
      "cycles_elapsed": 853675.0,
      "dram_read_mb": 451.653888,
      "dram_write_mb": 76.100864,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 92.989233,
      "tensor_pct": 92.989233,
      "dram_pct": 21.168798,
      "l2_pct": 47.743474,
      "warps_active_pct": 14.724952,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.960745795141008
     }
    }
   ],
   "dominant_kernel": "nvjet_sm90_tst_192x208_64x4_2x1_v_bz_coopB_bias_TNT",
   "primary": "compute_bound",
   "primary_rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
   "alternative_explanation": "Padding/redundant work inside nominal FLOPs; instruction mix vs tensor pipe",
   "counter_experiment": "Same dtype better tile/kernel; compare effective vs executed FLOPs",
   "status": "classified"
  },
  {
   "group": "dit:aten.linear.default:{'M': 2640, 'N': 5120, 'K': 13824}",
   "resolution": "704x1280",
   "stage": "dit",
   "op": "aten.linear.default",
   "shape_parameters": {
    "M": 2640,
    "N": 5120,
    "K": 13824
   },
   "input_sha256": "75bf41dc2a7a283f82bdcd0635998b9bc8c93a30b0ebb2b209ce44fd3b9e360c",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing704_1b9e93116d235408_v1",
   "scheduler_note": [],
   "nominal_mac_flops": 373712486400,
   "summed_kernel_us": 472.896,
   "kernels": [
    {
     "kernel": "nvjet_sm90_tst_192x192_64x4_1x2_h_bz_coopB_bias_TNN",
     "metrics": {
      "dur_us": 472.896,
      "sm_pct": 91.75556,
      "tensor_pct": 91.75556,
      "dram_pct": 20.160094,
      "l2_pct": 43.729417,
      "warps_active_pct": 14.576527,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 213.204,
      "smem_static_b": 0.0,
      "l2_hit_pct": 68.23287,
      "sm_cycles_active": 727671.189394,
      "cycles_elapsed": 782974.0,
      "dram_read_mb": 436.227072,
      "dram_write_mb": 22.518272,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 91.75556,
      "tensor_pct": 91.75556,
      "dram_pct": 20.160094,
      "l2_pct": 43.729417,
      "warps_active_pct": 14.576527,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.9293682668824252
     }
    }
   ],
   "dominant_kernel": "nvjet_sm90_tst_192x192_64x4_1x2_h_bz_coopB_bias_TNN",
   "primary": "compute_bound",
   "primary_rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
   "alternative_explanation": "Padding/redundant work inside nominal FLOPs; instruction mix vs tensor pipe",
   "counter_experiment": "Same dtype better tile/kernel; compare effective vs executed FLOPs",
   "status": "classified"
  },
  {
   "group": "dit:aten.linear.default:{'M': 2640, 'N': 13824, 'K': 5120}",
   "resolution": "704x1280",
   "stage": "dit",
   "op": "aten.linear.default",
   "shape_parameters": {
    "M": 2640,
    "N": 13824,
    "K": 5120
   },
   "input_sha256": "c08dc8cdb79c43328ef6849ee7e60049c20bbb243f50823bd9a80415eb91dc94",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing704_9ba227caa2b6dfb2_v1",
   "scheduler_note": [],
   "nominal_mac_flops": 373712486400,
   "summed_kernel_us": 451.03999999999996,
   "kernels": [
    {
     "kernel": "nvjet_sm90_tst_128x240_64x4_2x1_v_bz_coopA_bias_TNT",
     "metrics": {
      "dur_us": 451.04,
      "sm_pct": 94.507539,
      "tensor_pct": 94.507539,
      "dram_pct": 20.897931,
      "l2_pct": 53.350468,
      "warps_active_pct": 14.758237,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 204.964,
      "smem_static_b": 0.0,
      "l2_hit_pct": 72.172443,
      "sm_cycles_active": 719155.045455,
      "cycles_elapsed": 737859.0,
      "dram_read_mb": 384.699648,
      "dram_write_mb": 68.826368,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 94.507539,
      "tensor_pct": 94.507539,
      "dram_pct": 20.897931,
      "l2_pct": 53.350468,
      "warps_active_pct": 14.758237,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.9746510450573891
     }
    }
   ],
   "dominant_kernel": "nvjet_sm90_tst_128x240_64x4_2x1_v_bz_coopA_bias_TNT",
   "primary": "compute_bound",
   "primary_rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
   "alternative_explanation": "Padding/redundant work inside nominal FLOPs; instruction mix vs tensor pipe",
   "counter_experiment": "Same dtype better tile/kernel; compare effective vs executed FLOPs",
   "status": "classified"
  },
  {
   "group": "dit:aten.linear.default:{'M': 2640, 'N': 5120, 'K': 5120}",
   "resolution": "704x1280",
   "stage": "dit",
   "op": "aten.linear.default",
   "shape_parameters": {
    "M": 2640,
    "N": 5120,
    "K": 5120
   },
   "input_sha256": "984423718cf2b4425d4e6c3c3cc45ee71db6a1b3d36906b61da17257f41a950c",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing704_697ad043e70ad841_v1",
   "scheduler_note": [],
   "nominal_mac_flops": 138412032000,
   "summed_kernel_us": 183.52,
   "kernels": [
    {
     "kernel": "nvjet_sm90_tst_192x192_64x4_1x2_h_bz_coopB_bias_TNN",
     "metrics": {
      "dur_us": 183.52,
      "sm_pct": 88.174905,
      "tensor_pct": 88.174905,
      "dram_pct": 20.924745,
      "l2_pct": 42.676095,
      "warps_active_pct": 14.558241,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 213.204,
      "smem_static_b": 0.0,
      "l2_hit_pct": 69.070541,
      "sm_cycles_active": 279238.689394,
      "cycles_elapsed": 302800.0,
      "dram_read_mb": 161.599232,
      "dram_write_mb": 23.164672,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 88.174905,
      "tensor_pct": 88.174905,
      "dram_pct": 20.924745,
      "l2_pct": 42.676095,
      "warps_active_pct": 14.558241,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.9221885382892998
     }
    }
   ],
   "dominant_kernel": "nvjet_sm90_tst_192x192_64x4_1x2_h_bz_coopB_bias_TNN",
   "primary": "compute_bound",
   "primary_rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
   "alternative_explanation": "Padding/redundant work inside nominal FLOPs; instruction mix vs tensor pipe",
   "counter_experiment": "Same dtype better tile/kernel; compare effective vs executed FLOPs",
   "status": "classified"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 96, 6, 706, 322], 'weight': [96, 96, 3, 3, 3], 'output': [1, 96, 4, 704, 320], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "704x1280",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     96,
     6,
     706,
     322
    ],
    "weight": [
     96,
     96,
     3,
     3,
     3
    ],
    "output": [
     1,
     96,
     4,
     704,
     320
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "361110b0df425093397551c589c4c8f8e119f195c1b7c8ce7fe463e598338f6e",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing704_0551e0aa26bb7201_v1",
   "scheduler_note": [],
   "nominal_mac_flops": 448454983680,
   "summed_kernel_us": 3677.9519999999998,
   "kernels": [
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 0.316928,
      "sm_pct": 37.462203,
      "tensor_pct": 0.0,
      "dram_pct": 67.515515,
      "l2_pct": 80.918825,
      "warps_active_pct": 94.629991,
      "max_warps_pct": 100.0,
      "waves": 121.09,
      "grid": 127875.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 55.198153,
      "sm_cycles_active": 612591.5,
      "cycles_elapsed": 622232.0,
      "dram_read_mb": 0.523785,
      "dram_write_mb": 505.793792,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "dram_bandwidth_bound",
     "rule": "DRAM throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 37.462203,
      "tensor_pct": 0.0,
      "dram_pct": 67.515515,
      "l2_pct": 80.918825,
      "warps_active_pct": 94.629991,
      "waves": 121.09,
      "sm_active_over_elapsed": 0.9845065827536996
     }
    },
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 0.003712,
      "sm_pct": 11.825452,
      "tensor_pct": 0.0,
      "dram_pct": 5.664136,
      "l2_pct": 14.672777,
      "warps_active_pct": 25.471919,
      "max_warps_pct": 100.0,
      "waves": 0.27,
      "grid": 288.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 58.541005,
      "sm_cycles_active": 2898.083333,
      "cycles_elapsed": 7293.0,
      "dram_read_mb": 0.001004,
      "dram_write_mb": 0.000256,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 11.825452,
      "tensor_pct": 0.0,
      "dram_pct": 5.664136,
      "l2_pct": 14.672777,
      "warps_active_pct": 25.471919,
      "waves": 0.27,
      "sm_active_over_elapsed": 0.3973787649801179
     }
    },
    {
     "kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize256x32x32_warpgroupsize1x1x1_g1_execute_segment_k",
     "metrics": {
      "dur_us": 2.926656,
      "sm_pct": 31.552625,
      "tensor_pct": 31.552625,
      "dram_pct": 9.867081,
      "l2_pct": 62.728996,
      "warps_active_pct": 18.36999,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 231.424,
      "smem_static_b": 0.0,
      "l2_hit_pct": 89.781518,
      "sm_cycles_active": 5254113.674242,
      "cycles_elapsed": 5266499.0,
      "dram_read_mb": 1.049221,
      "dram_write_mb": 340.380928,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "l2_bandwidth_bound",
     "rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
     "evidence": {
      "sm_pct": 31.552625,
      "tensor_pct": 31.552625,
      "dram_pct": 9.867081,
      "l2_pct": 62.728996,
      "warps_active_pct": 18.36999,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.997648281000718
     }
    },
    {
     "kernel": "void cudnn::nhwcToNchwKernel<float, float, float, 1, 0, 0>(cudnn::nhwc2nchw_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 0.18816,
      "sm_pct": 40.001026,
      "tensor_pct": 0.0,
      "dram_pct": 74.2135,
      "l2_pct": 83.138163,
      "warps_active_pct": 93.315156,
      "max_warps_pct": 100.0,
      "waves": 80.0,
      "grid": 84480.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 51.544234,
      "sm_cycles_active": 359505.893939,
      "cycles_elapsed": 366169.0,
      "dram_read_mb": 0.346048,
      "dram_write_mb": 325.758976,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "dram_bandwidth_bound",
     "rule": "DRAM throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 40.001026,
      "tensor_pct": 0.0,
      "dram_pct": 74.2135,
      "l2_pct": 83.138163,
      "warps_active_pct": 93.315156,
      "waves": 80.0,
      "sm_active_over_elapsed": 0.9818031945331254
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 0.242496,
      "sm_pct": 59.75911,
      "tensor_pct": 2.152885,
      "dram_pct": 57.199903,
      "l2_pct": 64.354426,
      "warps_active_pct": 83.664485,
      "max_warps_pct": 100.0,
      "waves": 160.0,
      "grid": 337920.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 50.228668,
      "sm_cycles_active": 471565.924242,
      "cycles_elapsed": 476638.0,
      "dram_read_mb": 0.346046,
      "dram_write_mb": 321.334528,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "l2_bandwidth_bound",
     "rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
     "evidence": {
      "sm_pct": 59.75911,
      "tensor_pct": 2.152885,
      "dram_pct": 57.199903,
      "l2_pct": 64.354426,
      "warps_active_pct": 83.664485,
      "waves": 160.0,
      "sm_active_over_elapsed": 0.9893586416567709
     }
    }
   ],
   "dominant_kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize256x32x32_warpgroupsize1x1x1_g1_execute_segment_k",
   "primary": "l2_bandwidth_bound",
   "primary_rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
   "alternative_explanation": "Working set thrashing L2; DRAM counter under-reads write-back",
   "counter_experiment": "Blocking/reuse change; compare lts bytes",
   "status": "classified"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 192, 6, 354, 162], 'weight': [192, 192, 3, 3, 3], 'output': [1, 192, 4, 352, 160], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "704x1280",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     192,
     6,
     354,
     162
    ],
    "weight": [
     192,
     192,
     3,
     3,
     3
    ],
    "output": [
     1,
     192,
     4,
     352,
     160
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "836c7ee7ca56334d7997004c7d80dc3a7742322bc7befd6ad4a0f29a4ec8ad0f",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing704_1b34e3d0553bbca4_v1",
   "scheduler_note": [],
   "nominal_mac_flops": 448454983680,
   "summed_kernel_us": 2130.8480000000004,
   "kernels": [
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 0.154816,
      "sm_pct": 38.994795,
      "tensor_pct": 0.0,
      "dram_pct": 68.579423,
      "l2_pct": 83.039358,
      "warps_active_pct": 92.634988,
      "max_warps_pct": 100.0,
      "waves": 61.1,
      "grid": 64518.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 55.616334,
      "sm_cycles_active": 297834.992424,
      "cycles_elapsed": 301887.0,
      "dram_read_mb": 264.27776,
      "dram_write_mb": 246.554624,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "dram_bandwidth_bound",
     "rule": "DRAM throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 38.994795,
      "tensor_pct": 0.0,
      "dram_pct": 68.579423,
      "l2_pct": 83.039358,
      "warps_active_pct": 92.634988,
      "waves": 61.1,
      "sm_active_over_elapsed": 0.9865777341323078
     }
    },
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 0.004992,
      "sm_pct": 32.519013,
      "tensor_pct": 0.0,
      "dram_pct": 16.697735,
      "l2_pct": 39.4128,
      "warps_active_pct": 74.119453,
      "max_warps_pct": 100.0,
      "waves": 1.09,
      "grid": 1152.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 57.22984,
      "sm_cycles_active": 5250.560606,
      "cycles_elapsed": 9805.0,
      "dram_read_mb": 3.990016,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 32.519013,
      "tensor_pct": 0.0,
      "dram_pct": 16.697735,
      "l2_pct": 39.4128,
      "warps_active_pct": 74.119453,
      "waves": 1.09,
      "sm_active_over_elapsed": 0.5354982770015299
     }
    },
    {
     "kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
     "metrics": {
      "dur_us": 1.759712,
      "sm_pct": 70.298033,
      "tensor_pct": 70.298033,
      "dram_pct": 9.176507,
      "l2_pct": 65.803113,
      "warps_active_pct": 18.352035,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 231.424,
      "smem_static_b": 0.0,
      "l2_hit_pct": 92.322282,
      "sm_cycles_active": 3067982.242424,
      "cycles_elapsed": 3150262.0,
      "dram_read_mb": 609.446144,
      "dram_write_mb": 167.589632,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 70.298033,
      "tensor_pct": 70.298033,
      "dram_pct": 9.176507,
      "l2_pct": 65.803113,
      "warps_active_pct": 18.352035,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.9738816144257209
     }
    },
    {
     "kernel": "void cudnn::nhwcToNchwKernel<float, float, float, 1, 0, 0>(cudnn::nhwc2nchw_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 0.089984,
      "sm_pct": 39.593024,
      "tensor_pct": 0.0,
      "dram_pct": 75.325003,
      "l2_pct": 85.992377,
      "warps_active_pct": 92.996436,
      "max_warps_pct": 100.0,
      "waves": 40.0,
      "grid": 42240.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 51.7449,
      "sm_cycles_active": 168820.143939,
      "cycles_elapsed": 175393.0,
      "dram_read_mb": 173.03296,
      "dram_write_mb": 153.057536,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "dram_bandwidth_bound",
     "rule": "DRAM throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 39.593024,
      "tensor_pct": 0.0,
      "dram_pct": 75.325003,
      "l2_pct": 85.992377,
      "warps_active_pct": 92.996436,
      "waves": 40.0,
      "sm_active_over_elapsed": 0.9625249806947825
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 0.121344,
      "sm_pct": 60.091202,
      "tensor_pct": 2.164225,
      "dram_pct": 55.068871,
      "l2_pct": 63.772259,
      "warps_active_pct": 83.859226,
      "max_warps_pct": 100.0,
      "waves": 80.0,
      "grid": 168960.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 50.235974,
      "sm_cycles_active": 230991.333333,
      "cycles_elapsed": 237164.0,
      "dram_read_mb": 173.032192,
      "dram_write_mb": 148.393984,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 60.091202,
      "tensor_pct": 2.164225,
      "dram_pct": 55.068871,
      "l2_pct": 63.772259,
      "warps_active_pct": 83.859226,
      "waves": 80.0,
      "sm_active_over_elapsed": 0.97397300320875
     }
    }
   ],
   "dominant_kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
   "primary": "compute_bound",
   "primary_rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
   "alternative_explanation": "Padding/redundant work inside nominal FLOPs; instruction mix vs tensor pipe",
   "counter_experiment": "Same dtype better tile/kernel; compare effective vs executed FLOPs",
   "status": "classified"
  },
  {
   "group": "dit:aten.linear.default:{'M': 10560, 'N': 5120, 'K': 5120}",
   "resolution": "704x1280",
   "stage": "dit",
   "op": "aten.linear.default",
   "shape_parameters": {
    "M": 10560,
    "N": 5120,
    "K": 5120
   },
   "input_sha256": "70fed7084bfb383625de9935f9cc6ad7e398ccdf921a0e71276f5cf1bdbdcb0d",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing704_606cff6747219bac_v1",
   "scheduler_note": [],
   "nominal_mac_flops": 553648128000,
   "summed_kernel_us": 650.3359999999999,
   "kernels": [
    {
     "kernel": "nvjet_sm90_tst_256x128_64x4_1x2_h_bz_coopA_bias_TNT",
     "metrics": {
      "dur_us": 650.336,
      "sm_pct": 93.499099,
      "tensor_pct": 93.499099,
      "dram_pct": 21.66624,
      "l2_pct": 55.006145,
      "warps_active_pct": 14.779993,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 213.188,
      "smem_static_b": 0.0,
      "l2_hit_pct": 75.954249,
      "sm_cycles_active": 1083686.659091,
      "cycles_elapsed": 1119471.0,
      "dram_read_mb": 569.92512,
      "dram_write_mb": 108.10752,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 93.499099,
      "tensor_pct": 93.499099,
      "dram_pct": 21.66624,
      "l2_pct": 55.006145,
      "warps_active_pct": 14.779993,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.968034597672472
     }
    }
   ],
   "dominant_kernel": "nvjet_sm90_tst_256x128_64x4_1x2_h_bz_coopA_bias_TNT",
   "primary": "compute_bound",
   "primary_rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
   "alternative_explanation": "Padding/redundant work inside nominal FLOPs; instruction mix vs tensor pipe",
   "counter_experiment": "Same dtype better tile/kernel; compare effective vs executed FLOPs",
   "status": "classified"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 384, 4, 178, 82], 'weight': [384, 384, 3, 3, 3], 'output': [1, 384, 2, 176, 80], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "704x1280",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     384,
     4,
     178,
     82
    ],
    "weight": [
     384,
     384,
     3,
     3,
     3
    ],
    "output": [
     1,
     384,
     2,
     176,
     80
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "ae178ca2bbf227553643572af23755eb3ed51289f32398e11a61f88fb81ea3a1",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing704_ac3438e4ac0a9485_v1",
   "scheduler_note": [],
   "nominal_mac_flops": 224227491840,
   "summed_kernel_us": 1129.5360000000003,
   "kernels": [
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 0.049568,
      "sm_pct": 40.797062,
      "tensor_pct": 0.0,
      "dram_pct": 67.453876,
      "l2_pct": 82.632261,
      "warps_active_pct": 93.249602,
      "max_warps_pct": 100.0,
      "waves": 20.74,
      "grid": 21900.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 51.613654,
      "sm_cycles_active": 94851.409091,
      "cycles_elapsed": 97987.0,
      "dram_read_mb": 89.6896,
      "dram_write_mb": 71.156992,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "dram_bandwidth_bound",
     "rule": "DRAM throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 40.797062,
      "tensor_pct": 0.0,
      "dram_pct": 67.453876,
      "l2_pct": 82.632261,
      "warps_active_pct": 93.249602,
      "waves": 20.74,
      "sm_active_over_elapsed": 0.9679999294906466
     }
    },
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 0.010144,
      "sm_pct": 62.904102,
      "tensor_pct": 0.0,
      "dram_pct": 34.105876,
      "l2_pct": 69.333043,
      "warps_active_pct": 84.235024,
      "max_warps_pct": 100.0,
      "waves": 4.36,
      "grid": 4608.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 56.267413,
      "sm_cycles_active": 15569.378788,
      "cycles_elapsed": 19796.0,
      "dram_read_mb": 15.934464,
      "dram_write_mb": 0.611328,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 62.904102,
      "tensor_pct": 0.0,
      "dram_pct": 34.105876,
      "l2_pct": 69.333043,
      "warps_active_pct": 84.235024,
      "waves": 4.36,
      "sm_active_over_elapsed": 0.7864911491210346
     }
    },
    {
     "kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
     "metrics": {
      "dur_us": 1.01648,
      "sm_pct": 60.635004,
      "tensor_pct": 60.635004,
      "dram_pct": 6.582824,
      "l2_pct": 57.383586,
      "warps_active_pct": 17.334903,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 231.424,
      "smem_static_b": 0.0,
      "l2_hit_pct": 92.039463,
      "sm_cycles_active": 1646663.840909,
      "cycles_elapsed": 1840429.0,
      "dram_read_mb": 280.979456,
      "dram_write_mb": 41.000704,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 60.635004,
      "tensor_pct": 60.635004,
      "dram_pct": 6.582824,
      "l2_pct": 57.383586,
      "warps_active_pct": 17.334903,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.8947173951882957
     }
    },
    {
     "kernel": "void cudnn::nhwcToNchwKernel<float, float, float, 1, 0, 0>(cudnn::nhwc2nchw_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 0.022016,
      "sm_pct": 39.533865,
      "tensor_pct": 0.0,
      "dram_pct": 61.661659,
      "l2_pct": 83.086082,
      "warps_active_pct": 88.529565,
      "max_warps_pct": 100.0,
      "waves": 10.0,
      "grid": 10560.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 50.967376,
      "sm_cycles_active": 40452.984848,
      "cycles_elapsed": 43338.0,
      "dram_read_mb": 43.271168,
      "dram_write_mb": 21.8688,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "dram_bandwidth_bound",
     "rule": "DRAM throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 39.533865,
      "tensor_pct": 0.0,
      "dram_pct": 61.661659,
      "l2_pct": 83.086082,
      "warps_active_pct": 88.529565,
      "waves": 10.0,
      "sm_active_over_elapsed": 0.9334298963496239
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 0.031328,
      "sm_pct": 57.755959,
      "tensor_pct": 2.0766,
      "dram_pct": 40.419862,
      "l2_pct": 59.393079,
      "warps_active_pct": 80.931138,
      "max_warps_pct": 100.0,
      "waves": 20.0,
      "grid": 42240.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 50.914259,
      "sm_cycles_active": 56327.340909,
      "cycles_elapsed": 61788.0,
      "dram_read_mb": 43.27168,
      "dram_write_mb": 17.565184,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 57.755959,
      "tensor_pct": 2.0766,
      "dram_pct": 40.419862,
      "l2_pct": 59.393079,
      "warps_active_pct": 80.931138,
      "waves": 20.0,
      "sm_active_over_elapsed": 0.9116226598854146
     }
    }
   ],
   "dominant_kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
   "primary": "compute_bound",
   "primary_rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
   "alternative_explanation": "Padding/redundant work inside nominal FLOPs; instruction mix vs tensor pipe",
   "counter_experiment": "Same dtype better tile/kernel; compare effective vs executed FLOPs",
   "status": "classified"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 384, 3, 90, 42], 'weight': [384, 384, 3, 3, 3], 'output': [1, 384, 1, 88, 40], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "704x1280",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     384,
     3,
     90,
     42
    ],
    "weight": [
     384,
     384,
     3,
     3,
     3
    ],
    "output": [
     1,
     384,
     1,
     88,
     40
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "fd2ac936cea7b4d5d60612a6045f5c612e27536de2424aa111e4bc689fc1e795",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing704_a50d349fe3a8545f_v1",
   "scheduler_note": [],
   "nominal_mac_flops": 28028436480,
   "summed_kernel_us": 180.32,
   "kernels": [
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 10.432,
      "sm_pct": 38.363356,
      "tensor_pct": 0.0,
      "dram_pct": 37.9252,
      "l2_pct": 72.882164,
      "warps_active_pct": 81.69635,
      "max_warps_pct": 100.0,
      "waves": 4.03,
      "grid": 4260.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 57.698826,
      "sm_cycles_active": 16991.242424,
      "cycles_elapsed": 20542.0,
      "dram_read_mb": 17.428224,
      "dram_write_mb": 1.563648,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "l2_bandwidth_bound",
     "rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
     "evidence": {
      "sm_pct": 38.363356,
      "tensor_pct": 0.0,
      "dram_pct": 37.9252,
      "l2_pct": 72.882164,
      "warps_active_pct": 81.69635,
      "waves": 4.03,
      "sm_active_over_elapsed": 0.8271464523415442
     }
    },
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 10.272,
      "sm_pct": 61.807772,
      "tensor_pct": 0.0,
      "dram_pct": 33.868591,
      "l2_pct": 68.398037,
      "warps_active_pct": 84.348962,
      "max_warps_pct": 100.0,
      "waves": 4.36,
      "grid": 4608.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 56.284997,
      "sm_cycles_active": 15577.439394,
      "cycles_elapsed": 20136.0,
      "dram_read_mb": 15.934464,
      "dram_write_mb": 0.778752,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 61.807772,
      "tensor_pct": 0.0,
      "dram_pct": 33.868591,
      "l2_pct": 68.398037,
      "warps_active_pct": 84.348962,
      "waves": 4.36,
      "sm_active_over_elapsed": 0.7736114120977354
     }
    },
    {
     "kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
     "metrics": {
      "dur_us": 147.488,
      "sm_pct": 53.616532,
      "tensor_pct": 53.616532,
      "dram_pct": 5.270415,
      "l2_pct": 47.961199,
      "warps_active_pct": 11.166581,
      "max_warps_pct": 18.75,
      "waves": 0.85,
      "grid": 112.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 231.424,
      "smem_static_b": 0.0,
      "l2_hit_pct": 93.975193,
      "sm_cycles_active": 178116.537879,
      "cycles_elapsed": 263688.0,
      "dram_read_mb": 33.41184,
      "dram_write_mb": 3.982336,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 53.616532,
      "tensor_pct": 53.616532,
      "dram_pct": 5.270415,
      "l2_pct": 47.961199,
      "warps_active_pct": 11.166581,
      "waves": 0.85,
      "sm_active_over_elapsed": 0.6754821526918177
     }
    },
    {
     "kernel": "void cudnn::nhwcToNchwKernel<float, float, float, 1, 0, 0>(cudnn::nhwc2nchw_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 5.568,
      "sm_pct": 20.483167,
      "tensor_pct": 0.0,
      "dram_pct": 20.347906,
      "l2_pct": 45.328474,
      "warps_active_pct": 69.817103,
      "max_warps_pct": 100.0,
      "waves": 1.25,
      "grid": 1320.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 53.493788,
      "sm_cycles_active": 6773.75,
      "cycles_elapsed": 10932.0,
      "dram_read_mb": 5.42208,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 20.483167,
      "tensor_pct": 0.0,
      "dram_pct": 20.347906,
      "l2_pct": 45.328474,
      "warps_active_pct": 69.817103,
      "waves": 1.25,
      "sm_active_over_elapsed": 0.6196258690084157
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 6.56,
      "sm_pct": 35.163881,
      "tensor_pct": 1.245718,
      "dram_pct": 17.263871,
      "l2_pct": 38.049735,
      "warps_active_pct": 68.626808,
      "max_warps_pct": 100.0,
      "waves": 2.5,
      "grid": 5280.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 52.557094,
      "sm_cycles_active": 8834.765152,
      "cycles_elapsed": 12874.0,
      "dram_read_mb": 5.422592,
      "dram_write_mb": 0.000256,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 35.163881,
      "tensor_pct": 1.245718,
      "dram_pct": 17.263871,
      "l2_pct": 38.049735,
      "warps_active_pct": 68.626808,
      "waves": 2.5,
      "sm_active_over_elapsed": 0.6862486524778624
     }
    }
   ],
   "dominant_kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
   "primary": "partial_wave_tail",
   "primary_rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
   "alternative_explanation": "Small M for this rank shard; multi-kernel split",
   "counter_experiment": "Batch/tile change or larger per-rank M; compare waves and duration",
   "status": "classified"
  },
  {
   "group": "dit:aten.linear.default:{'M': 10560, 'N': 5120, 'K': 1536}",
   "resolution": "704x1280",
   "stage": "dit",
   "op": "aten.linear.default",
   "shape_parameters": {
    "M": 10560,
    "N": 5120,
    "K": 1536
   },
   "input_sha256": "94f48ff6a98bf87d9ac8185c5db13ca77c238c34547a0c86dd5cac7b7d5b326f",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing704_4cfc56167ab9dd35_v1",
   "scheduler_note": [
    {
     "gpu_id": 0,
     "pid": 2019660,
     "user": "unknown",
     "kind": "foreign",
     "holder": "zhoutaichang",
     "holder_account": "zhoutaichang",
     "command": "[No data]",
     "memory_mb": 2266,
     "first_seen": "2026-09-10T05:20:53.219014752Z",
     "last_seen": "2026-09-10T05:20:54.100960726Z",
     "sightings": 3,
     "warn_count": 0,
     "last_warned": null,
     "last_signal": null,
     "end_time": "2026-09-10T05:20:55.549497427Z",
     "resolution": "process exited"
    }
   ],
   "nominal_mac_flops": 166094438400,
   "summed_kernel_us": 208.928,
   "kernels": [
    {
     "kernel": "nvjet_sm90_tst_256x128_64x4_1x2_h_bz_coopA_bias_TNT",
     "metrics": {
      "dur_us": 208.928,
      "sm_pct": 86.043035,
      "tensor_pct": 86.043035,
      "dram_pct": 16.886047,
      "l2_pct": 54.194965,
      "warps_active_pct": 14.740247,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 213.188,
      "smem_static_b": 0.0,
      "l2_hit_pct": 90.47683,
      "sm_cycles_active": 352520.257576,
      "cycles_elapsed": 364783.0,
      "dram_read_mb": 77.00992,
      "dram_write_mb": 92.732928,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 86.043035,
      "tensor_pct": 86.043035,
      "dram_pct": 16.886047,
      "l2_pct": 54.194965,
      "warps_active_pct": 14.740247,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.9663834596897334
     }
    }
   ],
   "dominant_kernel": "nvjet_sm90_tst_256x128_64x4_1x2_h_bz_coopA_bias_TNT",
   "primary": "compute_bound",
   "primary_rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
   "alternative_explanation": "Padding/redundant work inside nominal FLOPs; instruction mix vs tensor pipe",
   "counter_experiment": "Same dtype better tile/kernel; compare effective vs executed FLOPs",
   "status": "classified"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 192, 4, 178, 82], 'weight': [384, 192, 3, 3, 3], 'output': [1, 384, 2, 176, 80], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "704x1280",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     192,
     4,
     178,
     82
    ],
    "weight": [
     384,
     192,
     3,
     3,
     3
    ],
    "output": [
     1,
     384,
     2,
     176,
     80
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "146fb336db70349e18bee5b6332d0ab63940d20d95e88e53d065f0d87a4673e8",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing704_053cda5cb0c0dcf6_v1",
   "scheduler_note": [],
   "nominal_mac_flops": 112113745920,
   "summed_kernel_us": 603.9679999999998,
   "kernels": [
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 24.416,
      "sm_pct": 41.627845,
      "tensor_pct": 0.0,
      "dram_pct": 60.371031,
      "l2_pct": 81.6219,
      "warps_active_pct": 90.34357,
      "max_warps_pct": 100.0,
      "waves": 10.37,
      "grid": 10950.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 50.711385,
      "sm_cycles_active": 43423.878788,
      "cycles_elapsed": 48188.0,
      "dram_read_mb": 44.848896,
      "dram_write_mb": 25.988096,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "dram_bandwidth_bound",
     "rule": "DRAM throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 41.627845,
      "tensor_pct": 0.0,
      "dram_pct": 60.371031,
      "l2_pct": 81.6219,
      "warps_active_pct": 90.34357,
      "waves": 10.37,
      "sm_active_over_elapsed": 0.9011346971860215
     }
    },
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 6.752,
      "sm_pct": 47.45892,
      "tensor_pct": 0.0,
      "dram_pct": 24.646973,
      "l2_pct": 54.025325,
      "warps_active_pct": 79.402757,
      "max_warps_pct": 100.0,
      "waves": 2.18,
      "grid": 2304.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 56.19949,
      "sm_cycles_active": 8799.712121,
      "cycles_elapsed": 13216.0,
      "dram_read_mb": 7.971328,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 47.45892,
      "tensor_pct": 0.0,
      "dram_pct": 24.646973,
      "l2_pct": 54.025325,
      "warps_active_pct": 79.402757,
      "waves": 2.18,
      "sm_active_over_elapsed": 0.6658377815526635
     }
    },
    {
     "kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
     "metrics": {
      "dur_us": 519.68,
      "sm_pct": 59.65408,
      "tensor_pct": 59.65408,
      "dram_pct": 4.809225,
      "l2_pct": 55.831597,
      "warps_active_pct": 17.350418,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 231.424,
      "smem_static_b": 0.0,
      "l2_hit_pct": 95.376196,
      "sm_cycles_active": 821892.666667,
      "cycles_elapsed": 936401.0,
      "dram_read_mb": 82.941952,
      "dram_write_mb": 37.313792,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 59.65408,
      "tensor_pct": 59.65408,
      "dram_pct": 4.809225,
      "l2_pct": 55.831597,
      "warps_active_pct": 17.350418,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.8777144264764776
     }
    },
    {
     "kernel": "void cudnn::nhwcToNchwKernel<float, float, float, 1, 0, 0>(cudnn::nhwc2nchw_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 22.4,
      "sm_pct": 38.760723,
      "tensor_pct": 0.0,
      "dram_pct": 60.568158,
      "l2_pct": 81.773657,
      "warps_active_pct": 89.061875,
      "max_warps_pct": 100.0,
      "waves": 10.0,
      "grid": 10560.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 50.982941,
      "sm_cycles_active": 39132.878788,
      "cycles_elapsed": 44167.0,
      "dram_read_mb": 43.271424,
      "dram_write_mb": 21.942528,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "dram_bandwidth_bound",
     "rule": "DRAM throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 38.760723,
      "tensor_pct": 0.0,
      "dram_pct": 60.568158,
      "l2_pct": 81.773657,
      "warps_active_pct": 89.061875,
      "waves": 10.0,
      "sm_active_over_elapsed": 0.8860207573074921
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 30.72,
      "sm_pct": 58.861714,
      "tensor_pct": 2.116686,
      "dram_pct": 41.254908,
      "l2_pct": 60.558228,
      "warps_active_pct": 78.51864,
      "max_warps_pct": 100.0,
      "waves": 20.0,
      "grid": 42240.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 50.511728,
      "sm_cycles_active": 57132.060606,
      "cycles_elapsed": 60644.0,
      "dram_read_mb": 43.272192,
      "dram_write_mb": 17.630208,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "l2_bandwidth_bound",
     "rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
     "evidence": {
      "sm_pct": 58.861714,
      "tensor_pct": 2.116686,
      "dram_pct": 41.254908,
      "l2_pct": 60.558228,
      "warps_active_pct": 78.51864,
      "waves": 20.0,
      "sm_active_over_elapsed": 0.9420892521271684
     }
    }
   ],
   "dominant_kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
   "primary": "unresolved_no_dominant_limiter",
   "primary_rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
   "alternative_explanation": "Mixed limiter or measurement window artifact",
   "counter_experiment": "Collect stall reasons (smsp__warp_issue_stalled_*) and l1tex/lts bytes",
   "status": "unresolved"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 384, 4, 176, 80], 'weight': [768, 384, 3, 1, 1], 'output': [1, 768, 2, 176, 80], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "704x1280",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     384,
     4,
     176,
     80
    ],
    "weight": [
     768,
     384,
     3,
     1,
     1
    ],
    "output": [
     1,
     768,
     2,
     176,
     80
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "05ade70ee900de950adaf23171001399a4d6cc88017bf23770bc74ef28c01d3b",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing704_4273bf5ee693569e_v1",
   "scheduler_note": [],
   "nominal_mac_flops": 49828331520,
   "summed_kernel_us": 299.13599999999997,
   "kernels": [
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 48.256,
      "sm_pct": 40.434914,
      "tensor_pct": 0.0,
      "dram_pct": 66.100578,
      "l2_pct": 79.763297,
      "warps_active_pct": 92.503063,
      "max_warps_pct": 100.0,
      "waves": 20.0,
      "grid": 21120.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 51.595359,
      "sm_cycles_active": 88630.659091,
      "cycles_elapsed": 95368.0,
      "dram_read_mb": 86.532352,
      "dram_write_mb": 66.80576,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "dram_bandwidth_bound",
     "rule": "DRAM throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 40.434914,
      "tensor_pct": 0.0,
      "dram_pct": 66.100578,
      "l2_pct": 79.763297,
      "warps_active_pct": 92.503063,
      "waves": 20.0,
      "sm_active_over_elapsed": 0.9293542812159215
     }
    },
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 15.2,
      "sm_pct": 82.436652,
      "tensor_pct": 0.0,
      "dram_pct": 4.856264,
      "l2_pct": 12.976548,
      "warps_active_pct": 85.396044,
      "max_warps_pct": 100.0,
      "waves": 8.73,
      "grid": 9216.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 66.510858,
      "sm_cycles_active": 25746.409091,
      "cycles_elapsed": 30018.0,
      "dram_read_mb": 3.547904,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 82.436652,
      "tensor_pct": 0.0,
      "dram_pct": 4.856264,
      "l2_pct": 12.976548,
      "warps_active_pct": 85.396044,
      "waves": 8.73,
      "sm_active_over_elapsed": 0.8576990169564929
     }
    },
    {
     "kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
     "metrics": {
      "dur_us": 129.952,
      "sm_pct": 88.980992,
      "tensor_pct": 88.980992,
      "dram_pct": 33.04372,
      "l2_pct": 73.52943,
      "warps_active_pct": 17.852009,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 231.424,
      "smem_static_b": 0.0,
      "l2_hit_pct": 81.409008,
      "sm_cycles_active": 200409.651515,
      "cycles_elapsed": 208574.0,
      "dram_read_mb": 133.88288,
      "dram_write_mb": 72.706048,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 88.980992,
      "tensor_pct": 88.980992,
      "dram_pct": 33.04372,
      "l2_pct": 73.52943,
      "warps_active_pct": 17.852009,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.9608563460210765
     }
    },
    {
     "kernel": "void cudnn::nhwcToNchwKernel<float, float, float, 1, 0, 0>(cudnn::nhwc2nchw_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 45.056,
      "sm_pct": 37.93825,
      "tensor_pct": 0.0,
      "dram_pct": 70.172358,
      "l2_pct": 84.680591,
      "warps_active_pct": 89.034572,
      "max_warps_pct": 100.0,
      "waves": 20.0,
      "grid": 21120.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 51.095958,
      "sm_cycles_active": 86077.340909,
      "cycles_elapsed": 89006.0,
      "dram_read_mb": 86.524928,
      "dram_write_mb": 65.519104,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "dram_bandwidth_bound",
     "rule": "DRAM throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 37.93825,
      "tensor_pct": 0.0,
      "dram_pct": 70.172358,
      "l2_pct": 84.680591,
      "warps_active_pct": 89.034572,
      "waves": 20.0,
      "sm_active_over_elapsed": 0.9670959363301351
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 60.672,
      "sm_pct": 59.418102,
      "tensor_pct": 2.138901,
      "dram_pct": 50.323983,
      "l2_pct": 62.25211,
      "warps_active_pct": 83.980021,
      "max_warps_pct": 100.0,
      "waves": 40.0,
      "grid": 84480.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 50.460172,
      "sm_cycles_active": 114400.840909,
      "cycles_elapsed": 119929.0,
      "dram_read_mb": 86.52672,
      "dram_write_mb": 60.369152,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "l2_bandwidth_bound",
     "rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
     "evidence": {
      "sm_pct": 59.418102,
      "tensor_pct": 2.138901,
      "dram_pct": 50.323983,
      "l2_pct": 62.25211,
      "warps_active_pct": 83.980021,
      "waves": 40.0,
      "sm_active_over_elapsed": 0.953904734542938
     }
    }
   ],
   "dominant_kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
   "primary": "compute_bound",
   "primary_rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
   "alternative_explanation": "Padding/redundant work inside nominal FLOPs; instruction mix vs tensor pipe",
   "counter_experiment": "Same dtype better tile/kernel; compare effective vs executed FLOPs",
   "status": "classified"
  },
  {
   "group": "dit:aten.linear.default:{'M': 10560, 'N': 5120, 'K': 144}",
   "resolution": "704x1280",
   "stage": "dit",
   "op": "aten.linear.default",
   "shape_parameters": {
    "M": 10560,
    "N": 5120,
    "K": 144
   },
   "input_sha256": "31fe6110c4fdd22ab61ff0c539192126b4bc8da9a7355f9916c23de05f620d74",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing704_3d9ee58b5b1e2e00_v1",
   "scheduler_note": [],
   "nominal_mac_flops": 15571353600,
   "summed_kernel_us": 53.472,
   "kernels": [
    {
     "kernel": "nvjet_sm90_tst_128x192_64x5_2x1_v_bz_coopB_bias_TNN",
     "metrics": {
      "dur_us": 53.472,
      "sm_pct": 40.236191,
      "tensor_pct": 40.236191,
      "dram_pct": 33.664381,
      "l2_pct": 67.208801,
      "warps_active_pct": 14.739979,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 221.42,
      "smem_static_b": 0.0,
      "l2_hit_pct": 98.083377,
      "sm_cycles_active": 89445.30303,
      "cycles_elapsed": 95897.0,
      "dram_read_mb": 4.755456,
      "dram_write_mb": 81.8496,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "l2_bandwidth_bound",
     "rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
     "evidence": {
      "sm_pct": 40.236191,
      "tensor_pct": 40.236191,
      "dram_pct": 33.664381,
      "l2_pct": 67.208801,
      "warps_active_pct": 14.739979,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.9327226402285785
     }
    }
   ],
   "dominant_kernel": "nvjet_sm90_tst_128x192_64x5_2x1_v_bz_coopB_bias_TNN",
   "primary": "l2_bandwidth_bound",
   "primary_rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
   "alternative_explanation": "Working set thrashing L2; DRAM counter under-reads write-back",
   "counter_experiment": "Blocking/reuse change; compare lts bytes",
   "status": "classified"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 96, 3, 706, 322], 'weight': [96, 96, 3, 3, 3], 'output': [1, 96, 1, 704, 320], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "704x1280",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     96,
     3,
     706,
     322
    ],
    "weight": [
     96,
     96,
     3,
     3,
     3
    ],
    "output": [
     1,
     96,
     1,
     704,
     320
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "85c0ba2d39aa4aa6ffbabaa237b9fa5d7721eefd855682c8cf482c661a5943d9",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing704_1ae7dbe8166a0b6b_v1",
   "scheduler_note": [],
   "nominal_mac_flops": 112113745920,
   "summed_kernel_us": 1006.016,
   "kernels": [
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 158.496,
      "sm_pct": 37.586965,
      "tensor_pct": 0.0,
      "dram_pct": 66.627572,
      "l2_pct": 81.834161,
      "warps_active_pct": 93.257434,
      "max_warps_pct": 100.0,
      "waves": 60.55,
      "grid": 63939.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 56.822727,
      "sm_cycles_active": 305336.007576,
      "cycles_elapsed": 310248.0,
      "dram_read_mb": 261.903104,
      "dram_write_mb": 246.129408,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "dram_bandwidth_bound",
     "rule": "DRAM throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 37.586965,
      "tensor_pct": 0.0,
      "dram_pct": 66.627572,
      "l2_pct": 81.834161,
      "warps_active_pct": 93.257434,
      "waves": 60.55,
      "sm_active_over_elapsed": 0.9841675291250871
     }
    },
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 3.488,
      "sm_pct": 12.603875,
      "tensor_pct": 0.0,
      "dram_pct": 6.065995,
      "l2_pct": 15.459878,
      "warps_active_pct": 24.031763,
      "max_warps_pct": 100.0,
      "waves": 0.27,
      "grid": 288.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 57.407244,
      "sm_cycles_active": 3039.659091,
      "cycles_elapsed": 6807.0,
      "dram_read_mb": 1.004288,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 12.603875,
      "tensor_pct": 0.0,
      "dram_pct": 6.065995,
      "l2_pct": 15.459878,
      "warps_active_pct": 24.031763,
      "waves": 0.27,
      "sm_active_over_elapsed": 0.4465490070515646
     }
    },
    {
     "kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize256x32x32_warpgroupsize1x1x1_g1_execute_segment_k",
     "metrics": {
      "dur_us": 738.976,
      "sm_pct": 31.264587,
      "tensor_pct": 31.264587,
      "dram_pct": 9.724301,
      "l2_pct": 59.885082,
      "warps_active_pct": 18.503275,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 231.424,
      "smem_static_b": 0.0,
      "l2_hit_pct": 88.679848,
      "sm_cycles_active": 1272400.007576,
      "cycles_elapsed": 1329652.0,
      "dram_read_mb": 262.976,
      "dram_write_mb": 82.790656,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 31.264587,
      "tensor_pct": 31.264587,
      "dram_pct": 9.724301,
      "l2_pct": 59.885082,
      "warps_active_pct": 18.503275,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.9569421228832807
     }
    },
    {
     "kernel": "void cudnn::nhwcToNchwKernel<float, float, float, 1, 0, 0>(cudnn::nhwc2nchw_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 44.384,
      "sm_pct": 41.746253,
      "tensor_pct": 0.0,
      "dram_pct": 70.914106,
      "l2_pct": 85.734401,
      "warps_active_pct": 91.685438,
      "max_warps_pct": 100.0,
      "waves": 20.0,
      "grid": 21120.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 51.084849,
      "sm_cycles_active": 82649.909091,
      "cycles_elapsed": 87675.0,
      "dram_read_mb": 86.537728,
      "dram_write_mb": 64.758528,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "dram_bandwidth_bound",
     "rule": "DRAM throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 41.746253,
      "tensor_pct": 0.0,
      "dram_pct": 70.914106,
      "l2_pct": 85.734401,
      "warps_active_pct": 91.685438,
      "waves": 20.0,
      "sm_active_over_elapsed": 0.9426850195722839
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 60.672,
      "sm_pct": 59.459378,
      "tensor_pct": 2.140278,
      "dram_pct": 51.000054,
      "l2_pct": 62.565379,
      "warps_active_pct": 84.007886,
      "max_warps_pct": 100.0,
      "waves": 40.0,
      "grid": 84480.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 50.394864,
      "sm_cycles_active": 114214.090909,
      "cycles_elapsed": 119891.0,
      "dram_read_mb": 86.523904,
      "dram_write_mb": 62.30656,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "l2_bandwidth_bound",
     "rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
     "evidence": {
      "sm_pct": 59.459378,
      "tensor_pct": 2.140278,
      "dram_pct": 51.000054,
      "l2_pct": 62.565379,
      "warps_active_pct": 84.007886,
      "waves": 40.0,
      "sm_active_over_elapsed": 0.9526494141261647
     }
    }
   ],
   "dominant_kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize256x32x32_warpgroupsize1x1x1_g1_execute_segment_k",
   "primary": "unresolved_no_dominant_limiter",
   "primary_rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
   "alternative_explanation": "Mixed limiter or measurement window artifact",
   "counter_experiment": "Collect stall reasons (smsp__warp_issue_stalled_*) and l1tex/lts bytes",
   "status": "unresolved"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 192, 3, 354, 162], 'weight': [192, 192, 3, 3, 3], 'output': [1, 192, 1, 352, 160], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "704x1280",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     192,
     3,
     354,
     162
    ],
    "weight": [
     192,
     192,
     3,
     3,
     3
    ],
    "output": [
     1,
     192,
     1,
     352,
     160
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "a9625d3e84c044a16224205940b2918538dcbebc7b9e36a51af82b718a774552",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing704_996e506f74aef428_v1",
   "scheduler_note": [],
   "nominal_mac_flops": 112113745920,
   "summed_kernel_us": 562.496,
   "kernels": [
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 77.312,
      "sm_pct": 38.599213,
      "tensor_pct": 0.0,
      "dram_pct": 66.201483,
      "l2_pct": 83.499739,
      "warps_active_pct": 94.606701,
      "max_warps_pct": 100.0,
      "waves": 30.55,
      "grid": 32262.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 57.174954,
      "sm_cycles_active": 145466.515152,
      "cycles_elapsed": 152477.0,
      "dram_read_mb": 132.146944,
      "dram_write_mb": 113.996032,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "dram_bandwidth_bound",
     "rule": "DRAM throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 38.599213,
      "tensor_pct": 0.0,
      "dram_pct": 66.201483,
      "l2_pct": 83.499739,
      "warps_active_pct": 94.606701,
      "waves": 30.55,
      "sm_active_over_elapsed": 0.9540226732687553
     }
    },
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 5.152,
      "sm_pct": 31.641973,
      "tensor_pct": 0.0,
      "dram_pct": 16.259436,
      "l2_pct": 38.61829,
      "warps_active_pct": 73.963813,
      "max_warps_pct": 100.0,
      "waves": 1.09,
      "grid": 1152.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 56.795213,
      "sm_cycles_active": 5170.083333,
      "cycles_elapsed": 10067.0,
      "dram_read_mb": 3.990272,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 31.641973,
      "tensor_pct": 0.0,
      "dram_pct": 16.259436,
      "l2_pct": 38.61829,
      "warps_active_pct": 73.963813,
      "waves": 1.09,
      "sm_active_over_elapsed": 0.5135674315088904
     }
    },
    {
     "kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
     "metrics": {
      "dur_us": 427.04,
      "sm_pct": 73.067093,
      "tensor_pct": 73.067093,
      "dram_pct": 8.585092,
      "l2_pct": 64.293498,
      "warps_active_pct": 16.744155,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 231.424,
      "smem_static_b": 0.0,
      "l2_hit_pct": 91.064722,
      "sm_cycles_active": 696410.75,
      "cycles_elapsed": 767284.0,
      "dram_read_mb": 138.944,
      "dram_write_mb": 37.46048,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 73.067093,
      "tensor_pct": 73.067093,
      "dram_pct": 8.585092,
      "l2_pct": 64.293498,
      "warps_active_pct": 16.744155,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.9076310075539175
     }
    },
    {
     "kernel": "void cudnn::nhwcToNchwKernel<float, float, float, 1, 0, 0>(cudnn::nhwc2nchw_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 22.464,
      "sm_pct": 39.39385,
      "tensor_pct": 0.0,
      "dram_pct": 62.209485,
      "l2_pct": 81.155405,
      "warps_active_pct": 89.162527,
      "max_warps_pct": 100.0,
      "waves": 10.0,
      "grid": 10560.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 50.840402,
      "sm_cycles_active": 38699.712121,
      "cycles_elapsed": 44280.0,
      "dram_read_mb": 43.270656,
      "dram_write_mb": 23.85024,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "dram_bandwidth_bound",
     "rule": "DRAM throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 39.39385,
      "tensor_pct": 0.0,
      "dram_pct": 62.209485,
      "l2_pct": 81.155405,
      "warps_active_pct": 89.162527,
      "waves": 10.0,
      "sm_active_over_elapsed": 0.8739772385049683
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 30.528,
      "sm_pct": 59.221936,
      "tensor_pct": 2.129322,
      "dram_pct": 41.731601,
      "l2_pct": 60.937512,
      "warps_active_pct": 80.401756,
      "max_warps_pct": 100.0,
      "waves": 20.0,
      "grid": 42240.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 50.668949,
      "sm_cycles_active": 55837.113636,
      "cycles_elapsed": 60266.0,
      "dram_read_mb": 43.270912,
      "dram_write_mb": 17.987072,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "l2_bandwidth_bound",
     "rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
     "evidence": {
      "sm_pct": 59.221936,
      "tensor_pct": 2.129322,
      "dram_pct": 41.731601,
      "l2_pct": 60.937512,
      "warps_active_pct": 80.401756,
      "waves": 20.0,
      "sm_active_over_elapsed": 0.9265110283742077
     }
    }
   ],
   "dominant_kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
   "primary": "compute_bound",
   "primary_rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
   "alternative_explanation": "Padding/redundant work inside nominal FLOPs; instruction mix vs tensor pipe",
   "counter_experiment": "Same dtype better tile/kernel; compare effective vs executed FLOPs",
   "status": "classified"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 384, 3, 178, 82], 'weight': [384, 384, 3, 3, 3], 'output': [1, 384, 1, 176, 80], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "704x1280",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     384,
     3,
     178,
     82
    ],
    "weight": [
     384,
     384,
     3,
     3,
     3
    ],
    "output": [
     1,
     384,
     1,
     176,
     80
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "9b2cc0033aa13a80bc2dd24c544a6b1097e94f10f8c1f3748cbca111210d5fa1",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing704_e70c8df9d1c257cc_v1",
   "scheduler_note": [],
   "nominal_mac_flops": 112113745920,
   "summed_kernel_us": 585.4079999999999,
   "kernels": [
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 37.088,
      "sm_pct": 40.939927,
      "tensor_pct": 0.0,
      "dram_pct": 64.810161,
      "l2_pct": 85.309148,
      "warps_active_pct": 90.636878,
      "max_warps_pct": 100.0,
      "waves": 15.56,
      "grid": 16428.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 56.919947,
      "sm_cycles_active": 69242.5,
      "cycles_elapsed": 73312.0,
      "dram_read_mb": 67.268864,
      "dram_write_mb": 48.30208,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "dram_bandwidth_bound",
     "rule": "DRAM throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 40.939927,
      "tensor_pct": 0.0,
      "dram_pct": 64.810161,
      "l2_pct": 85.309148,
      "warps_active_pct": 90.636878,
      "waves": 15.56,
      "sm_active_over_elapsed": 0.9444906700130947
     }
    },
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 10.528,
      "sm_pct": 60.389793,
      "tensor_pct": 0.0,
      "dram_pct": 33.309839,
      "l2_pct": 66.459544,
      "warps_active_pct": 86.205833,
      "max_warps_pct": 100.0,
      "waves": 4.36,
      "grid": 4608.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 56.137436,
      "sm_cycles_active": 15372.113636,
      "cycles_elapsed": 20610.0,
      "dram_read_mb": 15.934464,
      "dram_write_mb": 0.882432,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 60.389793,
      "tensor_pct": 0.0,
      "dram_pct": 33.309839,
      "l2_pct": 66.459544,
      "warps_active_pct": 86.205833,
      "waves": 4.36,
      "sm_active_over_elapsed": 0.7458570420184376
     }
    },
    {
     "kernel": "void cask_plugin__5x_cudnn::init_device_workspace_kernel<xmma__5x_cudnn::Warp_specialized_params_non_template<xmma__5x_c",
     "metrics": {
      "dur_us": 4.8,
      "sm_pct": 0.043349,
      "tensor_pct": 0.0,
      "dram_pct": 0.020058,
      "l2_pct": 0.337283,
      "warps_active_pct": 1.630035,
      "max_warps_pct": 50.0,
      "waves": 0.0,
      "grid": 1.0,
      "block": 1.0,
      "regs": 16.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 51.112327,
      "sm_cycles_active": 43.818182,
      "cycles_elapsed": 9458.0,
      "dram_read_mb": 0.004608,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 128.0,
      "occ_limit_smem": 32.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 0.043349,
      "tensor_pct": 0.0,
      "dram_pct": 0.020058,
      "l2_pct": 0.337283,
      "warps_active_pct": 1.630035,
      "waves": 0.0,
      "sm_active_over_elapsed": 0.004632922605201945
     }
    },
    {
     "kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
     "metrics": {
      "dur_us": 504.928,
      "sm_pct": 60.616992,
      "tensor_pct": 60.616992,
      "dram_pct": 8.561562,
      "l2_pct": 55.778751,
      "warps_active_pct": 16.314718,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 231.424,
      "smem_static_b": 0.0,
      "l2_hit_pct": 89.897378,
      "sm_cycles_active": 791642.780303,
      "cycles_elapsed": 916169.0,
      "dram_read_mb": 179.351552,
      "dram_write_mb": 28.662016,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 60.616992,
      "tensor_pct": 60.616992,
      "dram_pct": 8.561562,
      "l2_pct": 55.778751,
      "warps_active_pct": 16.314718,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.864079422358757
     }
    },
    {
     "kernel": "void cudnn::nhwcToNchwKernel<float, float, float, 1, 0, 0>(cudnn::nhwc2nchw_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 12.064,
      "sm_pct": 36.357604,
      "tensor_pct": 0.0,
      "dram_pct": 43.6705,
      "l2_pct": 74.050607,
      "warps_active_pct": 86.737283,
      "max_warps_pct": 100.0,
      "waves": 5.0,
      "grid": 5280.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 51.477504,
      "sm_cycles_active": 19271.30303,
      "cycles_elapsed": 23727.0,
      "dram_read_mb": 21.643776,
      "dram_write_mb": 3.651072,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "l2_bandwidth_bound",
     "rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
     "evidence": {
      "sm_pct": 36.357604,
      "tensor_pct": 0.0,
      "dram_pct": 43.6705,
      "l2_pct": 74.050607,
      "warps_active_pct": 86.737283,
      "waves": 5.0,
      "sm_active_over_elapsed": 0.8122098465882749
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 16.0,
      "sm_pct": 56.852889,
      "tensor_pct": 2.039525,
      "dram_pct": 28.834673,
      "l2_pct": 60.084634,
      "warps_active_pct": 77.665579,
      "max_warps_pct": 100.0,
      "waves": 10.0,
      "grid": 21120.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 51.721564,
      "sm_cycles_active": 27622.901515,
      "cycles_elapsed": 31500.0,
      "dram_read_mb": 21.643008,
      "dram_write_mb": 0.514816,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "l2_bandwidth_bound",
     "rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
     "evidence": {
      "sm_pct": 56.852889,
      "tensor_pct": 2.039525,
      "dram_pct": 28.834673,
      "l2_pct": 60.084634,
      "warps_active_pct": 77.665579,
      "waves": 10.0,
      "sm_active_over_elapsed": 0.8769175084126984
     }
    }
   ],
   "dominant_kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
   "primary": "compute_bound",
   "primary_rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
   "alternative_explanation": "Padding/redundant work inside nominal FLOPs; instruction mix vs tensor pipe",
   "counter_experiment": "Same dtype better tile/kernel; compare effective vs executed FLOPs",
   "status": "classified"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 96, 6, 706, 322], 'weight': [3, 96, 3, 3, 3], 'output': [1, 3, 4, 704, 320], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "704x1280",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     96,
     6,
     706,
     322
    ],
    "weight": [
     3,
     96,
     3,
     3,
     3
    ],
    "output": [
     1,
     3,
     4,
     704,
     320
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "c09536053b980c5e962fadc017011cff0cd6c94409f5c2270984dfec83c493a3",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing704_0e33e5013ffbea1c_v1",
   "scheduler_note": [],
   "nominal_mac_flops": 14014218240,
   "summed_kernel_us": 1515.04,
   "kernels": [
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 2>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 0.31568,
      "sm_pct": 37.622393,
      "tensor_pct": 0.626804,
      "dram_pct": 67.741384,
      "l2_pct": 81.452424,
      "warps_active_pct": 93.745447,
      "max_warps_pct": 100.0,
      "waves": 121.09,
      "grid": 127875.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 55.225595,
      "sm_cycles_active": 615923.704545,
      "cycles_elapsed": 619644.0,
      "dram_read_mb": 0.523789,
      "dram_write_mb": 505.148672,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "dram_bandwidth_bound",
     "rule": "DRAM throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 37.622393,
      "tensor_pct": 0.626804,
      "dram_pct": 67.741384,
      "l2_pct": 81.452424,
      "warps_active_pct": 93.745447,
      "waves": 121.09,
      "sm_active_over_elapsed": 0.9939960760452776
     }
    },
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 2>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 0.003296,
      "sm_pct": 0.356453,
      "tensor_pct": 0.004228,
      "dram_pct": 0.257799,
      "l2_pct": 1.149234,
      "warps_active_pct": 11.983411,
      "max_warps_pct": 100.0,
      "waves": 0.01,
      "grid": 9.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 51.984691,
      "sm_cycles_active": 186.840909,
      "cycles_elapsed": 6453.0,
      "dram_read_mb": 4e-05,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 0.356453,
      "tensor_pct": 0.004228,
      "dram_pct": 0.257799,
      "l2_pct": 1.149234,
      "warps_active_pct": 11.983411,
      "waves": 0.01,
      "sm_active_over_elapsed": 0.028954115760111577
     }
    },
    {
     "kernel": "sm80_xmma_fprop_implicit_gemm_indexed_wo_smem_tf32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x16x64_stage1_warpsize4x1x1_g",
     "metrics": {
      "dur_us": 1.148096,
      "sm_pct": 51.621077,
      "tensor_pct": 20.377272,
      "dram_pct": 22.10333,
      "l2_pct": 75.24804,
      "warps_active_pct": 30.370714,
      "max_warps_pct": 31.25,
      "waves": 10.67,
      "grid": 7040.0,
      "block": 128.0,
      "regs": 94.0,
      "smem_dyn_kb": 4.096,
      "smem_static_b": 0.0,
      "l2_hit_pct": 74.308587,
      "sm_cycles_active": 1985078.939394,
      "cycles_elapsed": 2064874.0,
      "dram_read_mb": 1.207012,
      "dram_write_mb": 14.110976,
      "occ_limit_regs": 5.0,
      "occ_limit_smem": 12.0
     },
     "classification": "l2_bandwidth_bound",
     "rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
     "evidence": {
      "sm_pct": 51.621077,
      "tensor_pct": 20.377272,
      "dram_pct": 22.10333,
      "l2_pct": 75.24804,
      "warps_active_pct": 30.370714,
      "waves": 10.67,
      "sm_active_over_elapsed": 0.9613559662206992
     }
    },
    {
     "kernel": "void cudnn::nhwcToNchwKernel<float, float, float, 1, 0, 0>(cudnn::nhwc2nchw_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 0.038368,
      "sm_pct": 81.301954,
      "tensor_pct": 0.0,
      "dram_pct": 5.885784,
      "l2_pct": 14.580013,
      "warps_active_pct": 87.339402,
      "max_warps_pct": 100.0,
      "waves": 26.67,
      "grid": 28160.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 64.812477,
      "sm_cycles_active": 71612.893939,
      "cycles_elapsed": 75775.0,
      "dram_read_mb": 0.010829,
      "dram_write_mb": 0.02048,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 81.301954,
      "tensor_pct": 0.0,
      "dram_pct": 5.885784,
      "l2_pct": 14.580013,
      "warps_active_pct": 87.339402,
      "waves": 26.67,
      "sm_active_over_elapsed": 0.945072833243154
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 0.0096,
      "sm_pct": 47.685405,
      "tensor_pct": 1.70295,
      "dram_pct": 23.523262,
      "l2_pct": 51.542645,
      "warps_active_pct": 74.09594,
      "max_warps_pct": 100.0,
      "waves": 5.0,
      "grid": 10560.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 51.927913,
      "sm_cycles_active": 15075.507576,
      "cycles_elapsed": 18849.0,
      "dram_read_mb": 0.010828,
      "dram_write_mb": 0.000768,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 47.685405,
      "tensor_pct": 1.70295,
      "dram_pct": 23.523262,
      "l2_pct": 51.542645,
      "warps_active_pct": 74.09594,
      "waves": 5.0,
      "sm_active_over_elapsed": 0.7998041050453605
     }
    }
   ],
   "dominant_kernel": "sm80_xmma_fprop_implicit_gemm_indexed_wo_smem_tf32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x16x64_stage1_warpsize4x1x1_g",
   "primary": "l2_bandwidth_bound",
   "primary_rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
   "alternative_explanation": "Working set thrashing L2; DRAM counter under-reads write-back",
   "counter_experiment": "Blocking/reuse change; compare lts bytes",
   "status": "classified"
  },
  {
   "group": "dit:aten.linear.default:{'M': 10560, 'N': 64, 'K': 5120}",
   "resolution": "704x1280",
   "stage": "dit",
   "op": "aten.linear.default",
   "shape_parameters": {
    "M": 10560,
    "N": 64,
    "K": 5120
   },
   "input_sha256": "26ebef86e4633eac87d2d256ccdab318b493bdf4c53c9c855e742e4fe4f4f537",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing704_5a8bc19ef82213ac_v1",
   "scheduler_note": [],
   "nominal_mac_flops": 6920601600,
   "summed_kernel_us": 31.135999999999996,
   "kernels": [
    {
     "kernel": "nvjet_sm90_tst_64x80_64x11_1x2_h_bz_bias_TNT",
     "metrics": {
      "dur_us": 31.136,
      "sm_pct": 22.954364,
      "tensor_pct": 22.954364,
      "dram_pct": 74.503733,
      "l2_pct": 77.028846,
      "warps_active_pct": 13.512909,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 219.484,
      "smem_static_b": 0.0,
      "l2_hit_pct": 21.811946,
      "sm_cycles_active": 49166.227273,
      "cycles_elapsed": 56142.0,
      "dram_read_mb": 108.811264,
      "dram_write_mb": 2.653696,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "dram_bandwidth_bound",
     "rule": "DRAM throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 22.954364,
      "tensor_pct": 22.954364,
      "dram_pct": 74.503733,
      "l2_pct": 77.028846,
      "warps_active_pct": 13.512909,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.8757476982116775
     }
    }
   ],
   "dominant_kernel": "nvjet_sm90_tst_64x80_64x11_1x2_h_bz_bias_TNT",
   "primary": "dram_bandwidth_bound",
   "primary_rule": "DRAM throughput >= 60% of peak sustained",
   "alternative_explanation": "Cache-unfriendly layout or spills inflating traffic",
   "counter_experiment": "Layout/fusion that lowers DRAM bytes; check bytes and time fall together",
   "status": "classified"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 384, 3, 88, 40], 'weight': [768, 384, 3, 1, 1], 'output': [1, 768, 1, 88, 40], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "704x1280",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     384,
     3,
     88,
     40
    ],
    "weight": [
     768,
     384,
     3,
     1,
     1
    ],
    "output": [
     1,
     768,
     1,
     88,
     40
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "9dd2254d46cabc12367da79ca54c5220c508380e8b7fd1b985050405f5f81ee3",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing704_fd5ad4e08e7ff9d8_v1",
   "scheduler_note": [],
   "nominal_mac_flops": 6228541440,
   "summed_kernel_us": 74.752,
   "kernels": [
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 9.856,
      "sm_pct": 37.814765,
      "tensor_pct": 0.0,
      "dram_pct": 36.660836,
      "l2_pct": 68.880476,
      "warps_active_pct": 82.425709,
      "max_warps_pct": 100.0,
      "waves": 3.75,
      "grid": 3960.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 51.454718,
      "sm_cycles_active": 15050.742424,
      "cycles_elapsed": 19387.0,
      "dram_read_mb": 16.229376,
      "dram_write_mb": 1.101568,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "l2_bandwidth_bound",
     "rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
     "evidence": {
      "sm_pct": 37.814765,
      "tensor_pct": 0.0,
      "dram_pct": 36.660836,
      "l2_pct": 68.880476,
      "warps_active_pct": 82.425709,
      "waves": 3.75,
      "sm_active_over_elapsed": 0.7763316874194047
     }
    },
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 15.168,
      "sm_pct": 82.72878,
      "tensor_pct": 0.0,
      "dram_pct": 4.874997,
      "l2_pct": 13.05391,
      "warps_active_pct": 85.259736,
      "max_warps_pct": 100.0,
      "waves": 8.73,
      "grid": 9216.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 67.145744,
      "sm_cycles_active": 25731.484848,
      "cycles_elapsed": 29910.0,
      "dram_read_mb": 3.547904,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 82.72878,
      "tensor_pct": 0.0,
      "dram_pct": 4.874997,
      "l2_pct": 13.05391,
      "warps_active_pct": 85.259736,
      "waves": 8.73,
      "sm_active_over_elapsed": 0.8602970527582748
     }
    },
    {
     "kernel": "void cask_plugin__5x_cudnn::init_device_workspace_kernel<xmma__5x_cudnn::Warp_specialized_params_non_template<xmma__5x_c",
     "metrics": {
      "dur_us": 4.384,
      "sm_pct": 0.039343,
      "tensor_pct": 0.0,
      "dram_pct": 0.0209,
      "l2_pct": 0.26054,
      "warps_active_pct": 1.678345,
      "max_warps_pct": 50.0,
      "waves": 0.0,
      "grid": 1.0,
      "block": 1.0,
      "regs": 16.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 47.791682,
      "sm_cycles_active": 37.704545,
      "cycles_elapsed": 8572.0,
      "dram_read_mb": 0.004352,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 128.0,
      "occ_limit_smem": 32.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 0.039343,
      "tensor_pct": 0.0,
      "dram_pct": 0.0209,
      "l2_pct": 0.26054,
      "warps_active_pct": 1.678345,
      "waves": 0.0,
      "sm_active_over_elapsed": 0.004398570345310313
     }
    },
    {
     "kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
     "metrics": {
      "dur_us": 27.584,
      "sm_pct": 46.849007,
      "tensor_pct": 46.849007,
      "dram_pct": 18.040615,
      "l2_pct": 52.334057,
      "warps_active_pct": 15.439062,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 231.424,
      "smem_static_b": 0.0,
      "l2_hit_pct": 78.093218,
      "sm_cycles_active": 40314.856061,
      "cycles_elapsed": 50266.0,
      "dram_read_mb": 20.38144,
      "dram_write_mb": 3.5456,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 46.849007,
      "tensor_pct": 46.849007,
      "dram_pct": 18.040615,
      "l2_pct": 52.334057,
      "warps_active_pct": 15.439062,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.802030319918036
     }
    },
    {
     "kernel": "void cudnn::nhwcToNchwKernel<float, float, float, 1, 0, 0>(cudnn::nhwc2nchw_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 8.128,
      "sm_pct": 27.067728,
      "tensor_pct": 0.0,
      "dram_pct": 27.769106,
      "l2_pct": 57.67761,
      "warps_active_pct": 82.181854,
      "max_warps_pct": 100.0,
      "waves": 2.5,
      "grid": 2640.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 52.106403,
      "sm_cycles_active": 10753.727273,
      "cycles_elapsed": 15991.0,
      "dram_read_mb": 10.8288,
      "dram_write_mb": 0.009984,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 27.067728,
      "tensor_pct": 0.0,
      "dram_pct": 27.769106,
      "l2_pct": 57.67761,
      "warps_active_pct": 82.181854,
      "waves": 2.5,
      "sm_active_over_elapsed": 0.6724862280657871
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 9.632,
      "sm_pct": 47.591335,
      "tensor_pct": 1.699822,
      "dram_pct": 23.496223,
      "l2_pct": 52.18756,
      "warps_active_pct": 74.70752,
      "max_warps_pct": 100.0,
      "waves": 5.0,
      "grid": 10560.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 52.760277,
      "sm_cycles_active": 15231.69697,
      "cycles_elapsed": 18900.0,
      "dram_read_mb": 10.830848,
      "dram_write_mb": 0.004352,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 47.591335,
      "tensor_pct": 1.699822,
      "dram_pct": 23.496223,
      "l2_pct": 52.18756,
      "warps_active_pct": 74.70752,
      "waves": 5.0,
      "sm_active_over_elapsed": 0.8059098925925926
     }
    }
   ],
   "dominant_kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
   "primary": "partial_wave_tail",
   "primary_rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
   "alternative_explanation": "Small M for this rank shard; multi-kernel split",
   "counter_experiment": "Batch/tile change or larger per-rank M; compare waves and duration",
   "status": "classified"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 192, 2, 176, 80], 'weight': [384, 192, 1, 1, 1], 'output': [1, 384, 2, 176, 80], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "704x1280",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     192,
     2,
     176,
     80
    ],
    "weight": [
     384,
     192,
     1,
     1,
     1
    ],
    "output": [
     1,
     384,
     2,
     176,
     80
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "bccd4b59806daff52b45f9c87ba32579441de0ed0adb3dd99356b7ebbcc1f4c7",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing704_9a2a65226b52659e_v1",
   "scheduler_note": [],
   "nominal_mac_flops": 4152360960,
   "summed_kernel_us": 80.96,
   "kernels": [
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::direct_copy_kernel_cuda(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 16.736,
      "sm_pct": 55.700949,
      "tensor_pct": 3.897292,
      "dram_pct": 31.109782,
      "l2_pct": 58.851907,
      "warps_active_pct": 80.743628,
      "max_warps_pct": 100.0,
      "waves": 10.0,
      "grid": 21120.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 51.862893,
      "sm_cycles_active": 28632.977273,
      "cycles_elapsed": 32969.0,
      "dram_read_mb": 21.641984,
      "dram_write_mb": 3.384576,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 55.700949,
      "tensor_pct": 3.897292,
      "dram_pct": 31.109782,
      "l2_pct": 58.851907,
      "warps_active_pct": 80.743628,
      "waves": 10.0,
      "sm_active_over_elapsed": 0.8684818245321363
     }
    },
    {
     "kernel": "void cutlass::Kernel2<cutlass_80_tensorop_s1688gemm_256x64_16x4_nn_align4>(Params)",
     "metrics": {
      "dur_us": 33.568,
      "sm_pct": 38.541084,
      "tensor_pct": 38.541084,
      "dram_pct": 25.477149,
      "l2_pct": 56.133692,
      "warps_active_pct": 11.155685,
      "max_warps_pct": 12.5,
      "waves": 2.55,
      "grid": 672.0,
      "block": 128.0,
      "regs": 214.0,
      "smem_dyn_kb": 81.92,
      "smem_static_b": 0.0,
      "l2_hit_pct": 81.513649,
      "sm_cycles_active": 55743.931818,
      "cycles_elapsed": 60023.0,
      "dram_read_mb": 22.0352,
      "dram_write_mb": 19.05152,
      "occ_limit_regs": 2.0,
      "occ_limit_smem": 2.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 38.541084,
      "tensor_pct": 38.541084,
      "dram_pct": 25.477149,
      "l2_pct": 56.133692,
      "warps_active_pct": 11.155685,
      "waves": 2.55,
      "sm_active_over_elapsed": 0.9287095249820901
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 30.656,
      "sm_pct": 58.936192,
      "tensor_pct": 2.119079,
      "dram_pct": 42.051514,
      "l2_pct": 60.877906,
      "warps_active_pct": 81.170483,
      "max_warps_pct": 100.0,
      "waves": 20.0,
      "grid": 42240.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 51.023367,
      "sm_cycles_active": 55555.871212,
      "cycles_elapsed": 60564.0,
      "dram_read_mb": 43.271168,
      "dram_write_mb": 18.736384,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "l2_bandwidth_bound",
     "rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
     "evidence": {
      "sm_pct": 58.936192,
      "tensor_pct": 2.119079,
      "dram_pct": 42.051514,
      "l2_pct": 60.877906,
      "warps_active_pct": 81.170483,
      "waves": 20.0,
      "sm_active_over_elapsed": 0.917308487088039
     }
    }
   ],
   "dominant_kernel": "void cutlass::Kernel2<cutlass_80_tensorop_s1688gemm_256x64_16x4_nn_align4>(Params)",
   "primary": "unresolved_no_dominant_limiter",
   "primary_rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
   "alternative_explanation": "Mixed limiter or measurement window artifact",
   "counter_experiment": "Collect stall reasons (smsp__warp_issue_stalled_*) and l1tex/lts bytes",
   "status": "unresolved"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 192, 3, 178, 82], 'weight': [384, 192, 3, 3, 3], 'output': [1, 384, 1, 176, 80], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "704x1280",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     192,
     3,
     178,
     82
    ],
    "weight": [
     384,
     192,
     3,
     3,
     3
    ],
    "output": [
     1,
     384,
     1,
     176,
     80
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "d56deba2cf2a7667416ca96682c1bcff892969b687097e1b3ea3921ac1a1d6ac",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing704_740d67537eaca4e1_v1",
   "scheduler_note": [],
   "nominal_mac_flops": 56056872960,
   "summed_kernel_us": 306.40000000000003,
   "kernels": [
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 18.976,
      "sm_pct": 40.26307,
      "tensor_pct": 0.0,
      "dram_pct": 54.591576,
      "l2_pct": 79.430109,
      "warps_active_pct": 87.141999,
      "max_warps_pct": 100.0,
      "waves": 7.78,
      "grid": 8214.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 57.065173,
      "sm_cycles_active": 32504.689394,
      "cycles_elapsed": 37416.0,
      "dram_read_mb": 33.639168,
      "dram_write_mb": 16.110336,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "l2_bandwidth_bound",
     "rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
     "evidence": {
      "sm_pct": 40.26307,
      "tensor_pct": 0.0,
      "dram_pct": 54.591576,
      "l2_pct": 79.430109,
      "warps_active_pct": 87.141999,
      "waves": 7.78,
      "sm_active_over_elapsed": 0.868737689598033
     }
    },
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 6.816,
      "sm_pct": 47.003145,
      "tensor_pct": 0.0,
      "dram_pct": 24.437219,
      "l2_pct": 53.921645,
      "warps_active_pct": 82.359237,
      "max_warps_pct": 100.0,
      "waves": 2.18,
      "grid": 2304.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 56.067067,
      "sm_cycles_active": 8699.545455,
      "cycles_elapsed": 13344.0,
      "dram_read_mb": 7.971584,
      "dram_write_mb": 0.000512,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 47.003145,
      "tensor_pct": 0.0,
      "dram_pct": 24.437219,
      "l2_pct": 53.921645,
      "warps_active_pct": 82.359237,
      "waves": 2.18,
      "sm_active_over_elapsed": 0.6519443536420862
     }
    },
    {
     "kernel": "void cask_plugin__5x_cudnn::init_device_workspace_kernel<xmma__5x_cudnn::Warp_specialized_params_non_template<xmma__5x_c",
     "metrics": {
      "dur_us": 4.96,
      "sm_pct": 0.042188,
      "tensor_pct": 0.0,
      "dram_pct": 0.020587,
      "l2_pct": 0.226394,
      "warps_active_pct": 1.57003,
      "max_warps_pct": 50.0,
      "waves": 0.0,
      "grid": 1.0,
      "block": 1.0,
      "regs": 16.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 51.091622,
      "sm_cycles_active": 44.015152,
      "cycles_elapsed": 9724.0,
      "dram_read_mb": 0.004864,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 128.0,
      "occ_limit_smem": 32.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 0.042188,
      "tensor_pct": 0.0,
      "dram_pct": 0.020587,
      "l2_pct": 0.226394,
      "warps_active_pct": 1.57003,
      "waves": 0.0,
      "sm_active_over_elapsed": 0.004526445084327437
     }
    },
    {
     "kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
     "metrics": {
      "dur_us": 247.264,
      "sm_pct": 62.617369,
      "tensor_pct": 62.617369,
      "dram_pct": 7.530364,
      "l2_pct": 56.794153,
      "warps_active_pct": 16.474858,
      "max_warps_pct": 18.75,
      "waves": 1.0,
      "grid": 132.0,
      "block": 384.0,
      "regs": 168.0,
      "smem_dyn_kb": 231.424,
      "smem_static_b": 0.0,
      "l2_hit_pct": 92.413696,
      "sm_cycles_active": 390520.234848,
      "cycles_elapsed": 445954.0,
      "dram_read_mb": 66.658816,
      "dram_write_mb": 22.921472,
      "occ_limit_regs": 1.0,
      "occ_limit_smem": 1.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 62.617369,
      "tensor_pct": 62.617369,
      "dram_pct": 7.530364,
      "l2_pct": 56.794153,
      "warps_active_pct": 16.474858,
      "waves": 1.0,
      "sm_active_over_elapsed": 0.8756962261757939
     }
    },
    {
     "kernel": "void cudnn::nhwcToNchwKernel<float, float, float, 1, 0, 0>(cudnn::nhwc2nchw_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 12.224,
      "sm_pct": 36.079744,
      "tensor_pct": 0.0,
      "dram_pct": 44.352836,
      "l2_pct": 72.918726,
      "warps_active_pct": 87.156686,
      "max_warps_pct": 100.0,
      "waves": 5.0,
      "grid": 5280.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 51.567496,
      "sm_cycles_active": 19012.833333,
      "cycles_elapsed": 23935.0,
      "dram_read_mb": 21.643776,
      "dram_write_mb": 4.352256,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "l2_bandwidth_bound",
     "rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
     "evidence": {
      "sm_pct": 36.079744,
      "tensor_pct": 0.0,
      "dram_pct": 44.352836,
      "l2_pct": 72.918726,
      "warps_active_pct": 87.156686,
      "waves": 5.0,
      "sm_active_over_elapsed": 0.7943527609358679
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 16.16,
      "sm_pct": 56.505241,
      "tensor_pct": 2.02731,
      "dram_pct": 28.769562,
      "l2_pct": 59.815566,
      "warps_active_pct": 76.979545,
      "max_warps_pct": 100.0,
      "waves": 10.0,
      "grid": 21120.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 51.482956,
      "sm_cycles_active": 27465.030303,
      "cycles_elapsed": 31711.0,
      "dram_read_mb": 21.642752,
      "dram_write_mb": 0.695808,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 56.505241,
      "tensor_pct": 2.02731,
      "dram_pct": 28.769562,
      "l2_pct": 59.815566,
      "warps_active_pct": 76.979545,
      "waves": 10.0,
      "sm_active_over_elapsed": 0.8661042005297847
     }
    }
   ],
   "dominant_kernel": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_",
   "primary": "compute_bound",
   "primary_rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
   "alternative_explanation": "Padding/redundant work inside nominal FLOPs; instruction mix vs tensor pipe",
   "counter_experiment": "Same dtype better tile/kernel; compare effective vs executed FLOPs",
   "status": "classified"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 16, 3, 90, 42], 'weight': [384, 16, 3, 3, 3], 'output': [1, 384, 1, 88, 40], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "704x1280",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     16,
     3,
     90,
     42
    ],
    "weight": [
     384,
     16,
     3,
     3,
     3
    ],
    "output": [
     1,
     384,
     1,
     88,
     40
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "f8fb1466dc2a1a6e997b5eff656e0d892480f377234df948eb6b06009d0b88b2",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing704_9e24c4a75ab83dea_v1",
   "scheduler_note": [],
   "nominal_mac_flops": 1167851520,
   "summed_kernel_us": 42.367999999999995,
   "kernels": [
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 2>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 3.328,
      "sm_pct": 11.366096,
      "tensor_pct": 0.163969,
      "dram_pct": 4.60672,
      "l2_pct": 12.750725,
      "warps_active_pct": 31.156783,
      "max_warps_pct": 100.0,
      "waves": 0.34,
      "grid": 355.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 61.933168,
      "sm_cycles_active": 2872.643939,
      "cycles_elapsed": 6565.0,
      "dram_read_mb": 0.734976,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 11.366096,
      "tensor_pct": 0.163969,
      "dram_pct": 4.60672,
      "l2_pct": 12.750725,
      "warps_active_pct": 31.156783,
      "waves": 0.34,
      "sm_active_over_elapsed": 0.43756952612338157
     }
    },
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 2>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 3.2,
      "sm_pct": 17.64436,
      "tensor_pct": 0.185585,
      "dram_pct": 4.405579,
      "l2_pct": 11.130917,
      "warps_active_pct": 33.868282,
      "max_warps_pct": 100.0,
      "waves": 0.36,
      "grid": 384.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 55.973597,
      "sm_cycles_active": 3164.022727,
      "cycles_elapsed": 6282.0,
      "dram_read_mb": 0.673024,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 17.64436,
      "tensor_pct": 0.185585,
      "dram_pct": 4.405579,
      "l2_pct": 11.130917,
      "warps_active_pct": 33.868282,
      "waves": 0.36,
      "sm_active_over_elapsed": 0.5036648721744668
     }
    },
    {
     "kernel": "sm80_xmma_fprop_implicit_gemm_tf32f32_tf32f32_f32_nhwckrsc_nchw_tilesize128x64x32_stage5_warpsize2x2x1_g1_tensor16x8x8_e",
     "metrics": {
      "dur_us": 29.408,
      "sm_pct": 24.811341,
      "tensor_pct": 24.811341,
      "dram_pct": 1.026599,
      "l2_pct": 19.082483,
      "warps_active_pct": 6.25421,
      "max_warps_pct": 6.25,
      "waves": 1.27,
      "grid": 168.0,
      "block": 128.0,
      "regs": 168.0,
      "smem_dyn_kb": 122.88,
      "smem_static_b": 0.0,
      "l2_hit_pct": 96.008118,
      "sm_cycles_active": 32763.310606,
      "cycles_elapsed": 55220.0,
      "dram_read_mb": 1.451776,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 3.0,
      "occ_limit_smem": 1.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 24.811341,
      "tensor_pct": 24.811341,
      "dram_pct": 1.026599,
      "l2_pct": 19.082483,
      "warps_active_pct": 6.25421,
      "waves": 1.27,
      "sm_active_over_elapsed": 0.5933232634190511
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 6.432,
      "sm_pct": 35.853635,
      "tensor_pct": 1.269382,
      "dram_pct": 17.604721,
      "l2_pct": 38.787897,
      "warps_active_pct": 72.671736,
      "max_warps_pct": 100.0,
      "waves": 2.5,
      "grid": 5280.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 53.092082,
      "sm_cycles_active": 9037.310606,
      "cycles_elapsed": 12634.0,
      "dram_read_mb": 5.422592,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 35.853635,
      "tensor_pct": 1.269382,
      "dram_pct": 17.604721,
      "l2_pct": 38.787897,
      "warps_active_pct": 72.671736,
      "waves": 2.5,
      "sm_active_over_elapsed": 0.7153166539496596
     }
    }
   ],
   "dominant_kernel": "sm80_xmma_fprop_implicit_gemm_tf32f32_tf32f32_f32_nhwckrsc_nchw_tilesize128x64x32_stage5_warpsize2x2x1_g1_tensor16x8x8_e",
   "primary": "partial_wave_tail",
   "primary_rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
   "alternative_explanation": "Small M for this rank shard; multi-kernel split",
   "counter_experiment": "Batch/tile change or larger per-rank M; compare waves and duration",
   "status": "classified"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 96, 3, 706, 322], 'weight': [3, 96, 3, 3, 3], 'output': [1, 3, 1, 704, 320], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "704x1280",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     96,
     3,
     706,
     322
    ],
    "weight": [
     3,
     96,
     3,
     3,
     3
    ],
    "output": [
     1,
     3,
     1,
     704,
     320
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "a2673addbb1775eb79e779d7635c4fbcf9dd34b0d4953bdf81e740bec14aa1ef",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing704_aa00674f1a72afcd_v1",
   "scheduler_note": [],
   "nominal_mac_flops": 3503554560,
   "summed_kernel_us": 502.5599999999999,
   "kernels": [
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 2>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 158.016,
      "sm_pct": 37.570242,
      "tensor_pct": 0.62567,
      "dram_pct": 66.475095,
      "l2_pct": 82.215211,
      "warps_active_pct": 94.813912,
      "max_warps_pct": 100.0,
      "waves": 60.55,
      "grid": 63939.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 56.796461,
      "sm_cycles_active": 303515.94697,
      "cycles_elapsed": 310411.0,
      "dram_read_mb": 261.910016,
      "dram_write_mb": 243.464192,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "dram_bandwidth_bound",
     "rule": "DRAM throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 37.570242,
      "tensor_pct": 0.62567,
      "dram_pct": 66.475095,
      "l2_pct": 82.215211,
      "warps_active_pct": 94.813912,
      "waves": 60.55,
      "sm_active_over_elapsed": 0.9777873431353914
     }
    },
    {
     "kernel": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 2>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 3.296,
      "sm_pct": 0.355851,
      "tensor_pct": 0.004221,
      "dram_pct": 0.257799,
      "l2_pct": 0.578387,
      "warps_active_pct": 14.966176,
      "max_warps_pct": 100.0,
      "waves": 0.01,
      "grid": 9.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 51.30798,
      "sm_cycles_active": 186.318182,
      "cycles_elapsed": 6467.0,
      "dram_read_mb": 0.040448,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 0.355851,
      "tensor_pct": 0.004221,
      "dram_pct": 0.257799,
      "l2_pct": 0.578387,
      "warps_active_pct": 14.966176,
      "waves": 0.01,
      "sm_active_over_elapsed": 0.028810604917272307
     }
    },
    {
     "kernel": "sm80_xmma_fprop_implicit_gemm_indexed_wo_smem_tf32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x16x64_stage1_warpsize4x1x1_g",
     "metrics": {
      "dur_us": 324.32,
      "sm_pct": 45.730038,
      "tensor_pct": 18.046653,
      "dram_pct": 18.153743,
      "l2_pct": 60.857539,
      "warps_active_pct": 28.397117,
      "max_warps_pct": 31.25,
      "waves": 2.67,
      "grid": 1760.0,
      "block": 128.0,
      "regs": 94.0,
      "smem_dyn_kb": 4.096,
      "smem_static_b": 0.0,
      "l2_hit_pct": 75.026411,
      "sm_cycles_active": 542980.795455,
      "cycles_elapsed": 583513.0,
      "dram_read_mb": 278.2208,
      "dram_write_mb": 5.054208,
      "occ_limit_regs": 5.0,
      "occ_limit_smem": 12.0
     },
     "classification": "l2_bandwidth_bound",
     "rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
     "evidence": {
      "sm_pct": 45.730038,
      "tensor_pct": 18.046653,
      "dram_pct": 18.153743,
      "l2_pct": 60.857539,
      "warps_active_pct": 28.397117,
      "waves": 2.67,
      "sm_active_over_elapsed": 0.9305376151945202
     }
    },
    {
     "kernel": "void cudnn::nhwcToNchwKernel<float, float, float, 1, 0, 0>(cudnn::nhwc2nchw_params_t<T3>, const T1 *, T2 *)",
     "metrics": {
      "dur_us": 12.224,
      "sm_pct": 64.279271,
      "tensor_pct": 0.0,
      "dram_pct": 4.637879,
      "l2_pct": 12.789653,
      "warps_active_pct": 84.5493,
      "max_warps_pct": 100.0,
      "waves": 6.67,
      "grid": 7040.0,
      "block": 256.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 4.224,
      "l2_hit_pct": 67.404035,
      "sm_cycles_active": 19038.909091,
      "cycles_elapsed": 24054.0,
      "dram_read_mb": 2.71872,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 8.0,
      "occ_limit_smem": 19.0
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 64.279271,
      "tensor_pct": 0.0,
      "dram_pct": 4.637879,
      "l2_pct": 12.789653,
      "warps_active_pct": 84.5493,
      "waves": 6.67,
      "sm_active_over_elapsed": 0.7915069880685126
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 4.704,
      "sm_pct": 24.949374,
      "tensor_pct": 0.867166,
      "dram_pct": 12.083042,
      "l2_pct": 26.574567,
      "warps_active_pct": 71.125525,
      "max_warps_pct": 100.0,
      "waves": 1.25,
      "grid": 2640.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 52.894905,
      "sm_cycles_active": 5448.227273,
      "cycles_elapsed": 9238.0,
      "dram_read_mb": 2.717696,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 24.949374,
      "tensor_pct": 0.867166,
      "dram_pct": 12.083042,
      "l2_pct": 26.574567,
      "warps_active_pct": 71.125525,
      "waves": 1.25,
      "sm_active_over_elapsed": 0.5897626405066032
     }
    }
   ],
   "dominant_kernel": "sm80_xmma_fprop_implicit_gemm_indexed_wo_smem_tf32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x16x64_stage1_warpsize4x1x1_g",
   "primary": "l2_bandwidth_bound",
   "primary_rule": "L2 (lts) throughput >= 60% while DRAM < 60%",
   "alternative_explanation": "Working set thrashing L2; DRAM counter under-reads write-back",
   "counter_experiment": "Blocking/reuse change; compare lts bytes",
   "status": "classified"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 192, 1, 176, 80], 'weight': [384, 192, 1, 1, 1], 'output': [1, 384, 1, 176, 80], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "704x1280",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     192,
     1,
     176,
     80
    ],
    "weight": [
     384,
     192,
     1,
     1,
     1
    ],
    "output": [
     1,
     384,
     1,
     176,
     80
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "85c1d12d0656fbd621625fe39cf33eef4823fe7d3338e672b380cbbe7e08c3bc",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing704_2ea2cf3e2f86821c_v1",
   "scheduler_note": [],
   "nominal_mac_flops": 2076180480,
   "summed_kernel_us": 36.704,
   "kernels": [
    {
     "kernel": "void cutlass::Kernel2<cutlass_80_tensorop_s1688gemm_128x64_16x6_nn_align4>(Params)",
     "metrics": {
      "dur_us": 20.48,
      "sm_pct": 39.212265,
      "tensor_pct": 31.589181,
      "dram_pct": 14.119686,
      "l2_pct": 46.797248,
      "warps_active_pct": 16.420292,
      "max_warps_pct": 18.75,
      "waves": 1.7,
      "grid": 672.0,
      "block": 128.0,
      "regs": 136.0,
      "smem_dyn_kb": 73.728,
      "smem_static_b": 0.0,
      "l2_hit_pct": 80.892876,
      "sm_cycles_active": 32188.106061,
      "cycles_elapsed": 36621.0,
      "dram_read_mb": 11.135744,
      "dram_write_mb": 2.754304,
      "occ_limit_regs": 3.0,
      "occ_limit_smem": 3.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 39.212265,
      "tensor_pct": 31.589181,
      "dram_pct": 14.119686,
      "l2_pct": 46.797248,
      "warps_active_pct": 16.420292,
      "waves": 1.7,
      "sm_active_over_elapsed": 0.8789521329565003
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 16.224,
      "sm_pct": 56.09709,
      "tensor_pct": 2.012437,
      "dram_pct": 29.31918,
      "l2_pct": 59.138082,
      "warps_active_pct": 77.342627,
      "max_warps_pct": 100.0,
      "waves": 10.0,
      "grid": 21120.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 51.170545,
      "sm_cycles_active": 27725.643939,
      "cycles_elapsed": 31937.0,
      "dram_read_mb": 21.643264,
      "dram_write_mb": 1.197312,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 56.09709,
      "tensor_pct": 2.012437,
      "dram_pct": 29.31918,
      "l2_pct": 59.138082,
      "warps_active_pct": 77.342627,
      "waves": 10.0,
      "sm_active_over_elapsed": 0.8681355148886871
     }
    }
   ],
   "dominant_kernel": "void cutlass::Kernel2<cutlass_80_tensorop_s1688gemm_128x64_16x6_nn_align4>(Params)",
   "primary": "unresolved_no_dominant_limiter",
   "primary_rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
   "alternative_explanation": "Mixed limiter or measurement window artifact",
   "counter_experiment": "Collect stall reasons (smsp__warp_issue_stalled_*) and l1tex/lts bytes",
   "status": "unresolved"
  },
  {
   "group": "vae:aten.conv3d.default:{'input': [1, 16, 3, 88, 160], 'weight': [16, 16, 1, 1, 1], 'output': [1, 16, 3, 88, 160], 'stride': [1, 1, 1], 'padding': [0, 0, 0], 'dilation': [1, 1, 1], 'groups': 1}",
   "resolution": "704x1280",
   "stage": "vae",
   "op": "aten.conv3d.default",
   "shape_parameters": {
    "input": [
     1,
     16,
     3,
     88,
     160
    ],
    "weight": [
     16,
     16,
     1,
     1,
     1
    ],
    "output": [
     1,
     16,
     3,
     88,
     160
    ],
    "stride": [
     1,
     1,
     1
    ],
    "padding": [
     0,
     0,
     0
    ],
    "dilation": [
     1,
     1,
     1
    ],
    "groups": 1
   },
   "input_sha256": "28dd946d4a53d1263bf5d552d2aacaa22bc7162bea85950320f6d0e02d5750e7",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing704_b611cd0d7be4ea16_v1",
   "scheduler_note": [],
   "nominal_mac_flops": 21626880,
   "summed_kernel_us": 10.495999999999999,
   "kernels": [
    {
     "kernel": "void cutlass::Kernel2<cutlass_80_tensorop_s1688gemm_128x64_16x6_nn_align4>(Params)",
     "metrics": {
      "dur_us": 5.824,
      "sm_pct": 23.516289,
      "tensor_pct": 4.475664,
      "dram_pct": 9.813669,
      "l2_pct": 24.316526,
      "warps_active_pct": 15.687418,
      "max_warps_pct": 18.75,
      "waves": 0.85,
      "grid": 336.0,
      "block": 128.0,
      "regs": 136.0,
      "smem_dyn_kb": 73.728,
      "smem_static_b": 0.0,
      "l2_hit_pct": 55.525416,
      "sm_cycles_active": 6727.0,
      "cycles_elapsed": 10754.0,
      "dram_read_mb": 2.730752,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 3.0,
      "occ_limit_smem": 3.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 23.516289,
      "tensor_pct": 4.475664,
      "dram_pct": 9.813669,
      "l2_pct": 24.316526,
      "warps_active_pct": 15.687418,
      "waves": 0.85,
      "sm_active_over_elapsed": 0.6255346847684582
     }
    },
    {
     "kernel": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &",
     "metrics": {
      "dur_us": 4.672,
      "sm_pct": 25.277386,
      "tensor_pct": 0.87878,
      "dram_pct": 12.240192,
      "l2_pct": 28.005679,
      "warps_active_pct": 71.326351,
      "max_warps_pct": 100.0,
      "waves": 1.25,
      "grid": 2640.0,
      "block": 128.0,
      "regs": 18.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 53.639311,
      "sm_cycles_active": 5452.181818,
      "cycles_elapsed": 9118.0,
      "dram_read_mb": 2.717696,
      "dram_write_mb": 0.0,
      "occ_limit_regs": 21.0,
      "occ_limit_smem": 32.0
     },
     "classification": "partial_wave_tail",
     "rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
     "evidence": {
      "sm_pct": 25.277386,
      "tensor_pct": 0.87878,
      "dram_pct": 12.240192,
      "l2_pct": 28.005679,
      "warps_active_pct": 71.326351,
      "waves": 1.25,
      "sm_active_over_elapsed": 0.5979580848870366
     }
    }
   ],
   "dominant_kernel": "void cutlass::Kernel2<cutlass_80_tensorop_s1688gemm_128x64_16x6_nn_align4>(Params)",
   "primary": "partial_wave_tail",
   "primary_rule": "SM active cycles < 85% of elapsed with <= 1.5 waves: grid too small for the machine (tail/launch-limited)",
   "alternative_explanation": "Small M for this rank shard; multi-kernel split",
   "counter_experiment": "Batch/tile change or larger per-rank M; compare waves and duration",
   "status": "classified"
  },
  {
   "group": "output:aten.to.dtype:{}",
   "resolution": "704x1280",
   "stage": "output",
   "op": "aten.to.dtype",
   "shape_parameters": {},
   "input_sha256": "80d26723a59d26f80a9dfde0de9bf8e28bb650baf110fa77f33200b5bb9e4198",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing704_59ef4c5b540ee17f_v1",
   "scheduler_note": [],
   "nominal_mac_flops": null,
   "summed_kernel_us": 64.63999999999999,
   "kernels": [
    {
     "kernel": "void unrolled_elementwise_kernel<direct_copy_kernel_cuda(TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() l",
     "metrics": {
      "dur_us": 64.64,
      "sm_pct": 57.807509,
      "tensor_pct": 0.0,
      "dram_pct": 37.370313,
      "l2_pct": 47.510617,
      "warps_active_pct": 90.044417,
      "max_warps_pct": 100.0,
      "waves": 22.5,
      "grid": 47520.0,
      "block": 128.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 21.308935,
      "sm_cycles_active": 122111.340909,
      "cycles_elapsed": 127855.0,
      "dram_read_mb": 97.357056,
      "dram_write_mb": 18.819328,
      "occ_limit_regs": 16.0,
      "occ_limit_smem": 32.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 57.807509,
      "tensor_pct": 0.0,
      "dram_pct": 37.370313,
      "l2_pct": 47.510617,
      "warps_active_pct": 90.044417,
      "waves": 22.5,
      "sm_active_over_elapsed": 0.9550767737593368
     }
    }
   ],
   "dominant_kernel": "void unrolled_elementwise_kernel<direct_copy_kernel_cuda(TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() l",
   "primary": "unresolved_no_dominant_limiter",
   "primary_rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
   "alternative_explanation": "Mixed limiter or measurement window artifact",
   "counter_experiment": "Collect stall reasons (smsp__warp_issue_stalled_*) and l1tex/lts bytes",
   "status": "unresolved"
  },
  {
   "group": "output:aten.to.dtype:{}",
   "resolution": "704x1280",
   "stage": "output",
   "op": "aten.to.dtype",
   "shape_parameters": {},
   "input_sha256": "ff3ca616760a5d842212457937cd00ab117a7e2049cc892ea9847400317b91f6",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_missing704_99c188aeb0ac98c3_v1",
   "scheduler_note": [],
   "nominal_mac_flops": null,
   "summed_kernel_us": 84.224,
   "kernels": [
    {
     "kernel": "void unrolled_elementwise_kernel<direct_copy_kernel_cuda(TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() l",
     "metrics": {
      "dur_us": 84.224,
      "sm_pct": 59.270813,
      "tensor_pct": 0.0,
      "dram_pct": 38.833031,
      "l2_pct": 49.213573,
      "warps_active_pct": 89.787204,
      "max_warps_pct": 100.0,
      "waves": 30.0,
      "grid": 63360.0,
      "block": 128.0,
      "regs": 32.0,
      "smem_dyn_kb": 0.0,
      "smem_static_b": 0.0,
      "l2_hit_pct": 21.13599,
      "sm_cycles_active": 162807.962121,
      "cycles_elapsed": 166366.0,
      "dram_read_mb": 129.7984,
      "dram_write_mb": 27.542784,
      "occ_limit_regs": 16.0,
      "occ_limit_smem": 32.0
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 59.270813,
      "tensor_pct": 0.0,
      "dram_pct": 38.833031,
      "l2_pct": 49.213573,
      "warps_active_pct": 89.787204,
      "waves": 30.0,
      "sm_active_over_elapsed": 0.9786131909224239
     }
    }
   ],
   "dominant_kernel": "void unrolled_elementwise_kernel<direct_copy_kernel_cuda(TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() l",
   "primary": "unresolved_no_dominant_limiter",
   "primary_rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
   "alternative_explanation": "Mixed limiter or measurement window artifact",
   "counter_experiment": "Collect stall reasons (smsp__warp_issue_stalled_*) and l1tex/lts bytes",
   "status": "unresolved"
  },
  {
   "group": "attention:480_short:fa2:self",
   "resolution": "480x832",
   "stage": "attention",
   "op": "flash_attn_v2",
   "shape_parameters": "{\"role\": \"self\", \"valid_keys\": 4680, \"allocated_keys\": 29640}",
   "input_sha256": "e86fd1d5515ffb6e8e8ba1c4a38e0b54b23bd7d8d4d931434a89f116a2186b96",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/attention_hardware_extension_v1/480_short/fa2",
   "kernels": [
    {
     "kernel": "void flash_fwd_kernel<Flash_fwd_kernel_traits<128, 128, 64, 4, 0, 0, bfloat16_t, Flash_kernel_traits<128, 128, 64, 4, bf",
     "metrics": {
      "dur_us": 332.48,
      "sm_pct": 54.1837,
      "tensor_pct": null,
      "dram_pct": 2.580393,
      "l2_pct": 18.019752,
      "warps_active_pct": 10.632445,
      "max_warps_pct": null,
      "waves": null,
      "grid": null,
      "block": null,
      "regs": null,
      "smem_dyn_kb": null,
      "smem_static_b": null,
      "l2_hit_pct": null,
      "sm_cycles_active": null,
      "cycles_elapsed": null,
      "dram_read_mb": 36.198656,
      "dram_write_mb": 5.081088,
      "occ_limit_regs": null,
      "occ_limit_smem": null
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 54.1837,
      "tensor_pct": null,
      "dram_pct": 2.580393,
      "l2_pct": 18.019752,
      "warps_active_pct": 10.632445,
      "waves": null,
      "sm_active_over_elapsed": null
     }
    }
   ],
   "dominant_kernel": "void flash_fwd_kernel<Flash_fwd_kernel_traits<128, 128, 64, 4, 0, 0, bfloat16_t, Flash_kernel_traits<128, 128, 64, 4, bf",
   "primary": "unresolved_no_dominant_limiter",
   "primary_rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
   "alternative_explanation": "Mixed limiter or measurement window artifact",
   "counter_experiment": "Collect stall reasons (smsp__warp_issue_stalled_*) and l1tex/lts bytes",
   "status": "unresolved"
  },
  {
   "group": "attention:480_short:fa3:self",
   "resolution": "480x832",
   "stage": "attention",
   "op": "flash_attn_v3",
   "shape_parameters": "{\"role\": \"self\", \"valid_keys\": 4680, \"allocated_keys\": 29640}",
   "input_sha256": "e86fd1d5515ffb6e8e8ba1c4a38e0b54b23bd7d8d4d931434a89f116a2186b96",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/attention_hardware_extension_v1/480_short/fa3",
   "kernels": [
    {
     "kernel": "void prepare_varlen_num_blocks_kernel<1, 0>(int, int, int, const int *, const int *, const int *, const int *, const int",
     "metrics": {
      "dur_us": 2.976,
      "sm_pct": 0.005492,
      "tensor_pct": null,
      "dram_pct": 0.049072,
      "l2_pct": 0.369433,
      "warps_active_pct": 1.565629,
      "max_warps_pct": null,
      "waves": null,
      "grid": null,
      "block": null,
      "regs": null,
      "smem_dyn_kb": null,
      "smem_static_b": null,
      "l2_hit_pct": null,
      "sm_cycles_active": null,
      "cycles_elapsed": null,
      "dram_read_mb": 0.006912,
      "dram_write_mb": 0.0,
      "occ_limit_regs": null,
      "occ_limit_smem": null
     },
     "classification": "latency_or_occupancy_limited",
     "rule": "all throughputs < 40% with low achieved occupancy or < 1 wave",
     "evidence": {
      "sm_pct": 0.005492,
      "tensor_pct": null,
      "dram_pct": 0.049072,
      "l2_pct": 0.369433,
      "warps_active_pct": 1.565629,
      "waves": null,
      "sm_active_over_elapsed": null
     }
    },
    {
     "kernel": "void device_kernel<enable_sm90_or_later<FlashAttnFwdSm90<CollectiveMainloopFwdSm90<2, cute::tuple<cute::C<1>, cute::C<1>",
     "metrics": {
      "dur_us": 176.768,
      "sm_pct": 72.757391,
      "tensor_pct": null,
      "dram_pct": 4.846778,
      "l2_pct": 28.317732,
      "warps_active_pct": 18.720104,
      "max_warps_pct": null,
      "waves": null,
      "grid": null,
      "block": null,
      "regs": null,
      "smem_dyn_kb": null,
      "smem_static_b": null,
      "l2_hit_pct": null,
      "sm_cycles_active": null,
      "cycles_elapsed": null,
      "dram_read_mb": 36.64,
      "dram_write_mb": 4.582912,
      "occ_limit_regs": null,
      "occ_limit_smem": null
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 72.757391,
      "tensor_pct": null,
      "dram_pct": 4.846778,
      "l2_pct": 28.317732,
      "warps_active_pct": 18.720104,
      "waves": null,
      "sm_active_over_elapsed": null
     }
    }
   ],
   "dominant_kernel": "void device_kernel<enable_sm90_or_later<FlashAttnFwdSm90<CollectiveMainloopFwdSm90<2, cute::tuple<cute::C<1>, cute::C<1>",
   "primary": "compute_bound",
   "primary_rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
   "alternative_explanation": "Padding/redundant work inside nominal FLOPs; instruction mix vs tensor pipe",
   "counter_experiment": "Same dtype better tile/kernel; compare effective vs executed FLOPs",
   "status": "classified"
  },
  {
   "group": "attention:704_short:fa2:self",
   "resolution": "704x1280",
   "stage": "attention",
   "op": "flash_attn_v2",
   "shape_parameters": "{\"role\": \"self\", \"valid_keys\": 10560, \"allocated_keys\": 66880}",
   "input_sha256": "468353cb47a9d93b52d299d1edd226524978db730a64597becca38700ad6c050",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/attention_hardware_extension_v1/704_short/fa2",
   "kernels": [
    {
     "kernel": "void flash_fwd_kernel<Flash_fwd_kernel_traits<128, 128, 64, 4, 0, 0, bfloat16_t, Flash_kernel_traits<128, 128, 64, 4, bf",
     "metrics": {
      "dur_us": 1594.7839999999999,
      "sm_pct": 55.544556,
      "tensor_pct": null,
      "dram_pct": 1.441039,
      "l2_pct": 22.594491,
      "warps_active_pct": 11.923779,
      "max_warps_pct": null,
      "waves": null,
      "grid": null,
      "block": null,
      "regs": null,
      "smem_dyn_kb": null,
      "smem_static_b": null,
      "l2_hit_pct": null,
      "sm_cycles_active": null,
      "cycles_elapsed": null,
      "dram_read_mb": 85.747968,
      "dram_write_mb": 24.83712,
      "occ_limit_regs": null,
      "occ_limit_smem": null
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 55.544556,
      "tensor_pct": null,
      "dram_pct": 1.441039,
      "l2_pct": 22.594491,
      "warps_active_pct": 11.923779,
      "waves": null,
      "sm_active_over_elapsed": null
     }
    }
   ],
   "dominant_kernel": "void flash_fwd_kernel<Flash_fwd_kernel_traits<128, 128, 64, 4, 0, 0, bfloat16_t, Flash_kernel_traits<128, 128, 64, 4, bf",
   "primary": "unresolved_no_dominant_limiter",
   "primary_rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
   "alternative_explanation": "Mixed limiter or measurement window artifact",
   "counter_experiment": "Collect stall reasons (smsp__warp_issue_stalled_*) and l1tex/lts bytes",
   "status": "unresolved"
  },
  {
   "group": "attention:704_short:fa3:self",
   "resolution": "704x1280",
   "stage": "attention",
   "op": "flash_attn_v3",
   "shape_parameters": "{\"role\": \"self\", \"valid_keys\": 10560, \"allocated_keys\": 66880}",
   "input_sha256": "468353cb47a9d93b52d299d1edd226524978db730a64597becca38700ad6c050",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/attention_hardware_extension_v1/704_short/fa3",
   "kernels": [
    {
     "kernel": "void prepare_varlen_num_blocks_kernel<1, 0>(int, int, int, const int *, const int *, const int *, const int *, const int",
     "metrics": {
      "dur_us": 2.944,
      "sm_pct": 0.005544,
      "tensor_pct": null,
      "dram_pct": 0.051328,
      "l2_pct": 0.372938,
      "warps_active_pct": 1.817082,
      "max_warps_pct": null,
      "waves": null,
      "grid": null,
      "block": null,
      "regs": null,
      "smem_dyn_kb": null,
      "smem_static_b": null,
      "l2_hit_pct": null,
      "sm_cycles_active": null,
      "cycles_elapsed": null,
      "dram_read_mb": 0.007168,
      "dram_write_mb": 0.0,
      "occ_limit_regs": null,
      "occ_limit_smem": null
     },
     "classification": "latency_or_occupancy_limited",
     "rule": "all throughputs < 40% with low achieved occupancy or < 1 wave",
     "evidence": {
      "sm_pct": 0.005544,
      "tensor_pct": null,
      "dram_pct": 0.051328,
      "l2_pct": 0.372938,
      "warps_active_pct": 1.817082,
      "waves": null,
      "sm_active_over_elapsed": null
     }
    },
    {
     "kernel": "void device_kernel<enable_sm90_or_later<FlashAttnFwdSm90<CollectiveMainloopFwdSm90<2, cute::tuple<cute::C<1>, cute::C<1>",
     "metrics": {
      "dur_us": 818.24,
      "sm_pct": 75.16756,
      "tensor_pct": null,
      "dram_pct": 2.610327,
      "l2_pct": 23.927212,
      "warps_active_pct": 18.750561,
      "max_warps_pct": null,
      "waves": null,
      "grid": null,
      "block": null,
      "regs": null,
      "smem_dyn_kb": null,
      "smem_static_b": null,
      "l2_hit_pct": null,
      "sm_cycles_active": null,
      "cycles_elapsed": null,
      "dram_read_mb": 81.78944,
      "dram_write_mb": 20.986624,
      "occ_limit_regs": null,
      "occ_limit_smem": null
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 75.16756,
      "tensor_pct": null,
      "dram_pct": 2.610327,
      "l2_pct": 23.927212,
      "warps_active_pct": 18.750561,
      "waves": null,
      "sm_active_over_elapsed": null
     }
    }
   ],
   "dominant_kernel": "void device_kernel<enable_sm90_or_later<FlashAttnFwdSm90<CollectiveMainloopFwdSm90<2, cute::tuple<cute::C<1>, cute::C<1>",
   "primary": "compute_bound",
   "primary_rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
   "alternative_explanation": "Padding/redundant work inside nominal FLOPs; instruction mix vs tensor pipe",
   "counter_experiment": "Same dtype better tile/kernel; compare effective vs executed FLOPs",
   "status": "classified"
  },
  {
   "group": "attention:704_long:fa2:self",
   "resolution": "704x1280",
   "stage": "attention",
   "op": "flash_attn_v2",
   "shape_parameters": "{\"role\": \"self\", \"valid_keys\": 63360, \"allocated_keys\": 66880}",
   "input_sha256": "8ab8d8e91a6d5757a64db4da0bf7d6b02a4ab0fb6c1bc2ca2c2524f9a1e2f113",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/attention_hardware_extension_v1/704_long/fa2",
   "kernels": [
    {
     "kernel": "void flash_fwd_kernel<Flash_fwd_kernel_traits<128, 128, 64, 4, 0, 0, bfloat16_t, Flash_kernel_traits<128, 128, 64, 4, bf",
     "metrics": {
      "dur_us": 9490.591999999999,
      "sm_pct": 55.830852,
      "tensor_pct": null,
      "dram_pct": 1.216341,
      "l2_pct": 24.511834,
      "warps_active_pct": 11.919707,
      "max_warps_pct": null,
      "waves": null,
      "grid": null,
      "block": null,
      "regs": null,
      "smem_dyn_kb": null,
      "smem_static_b": null,
      "l2_hit_pct": null,
      "sm_cycles_active": null,
      "cycles_elapsed": null,
      "dram_read_mb": 518.881792,
      "dram_write_mb": 36.608512,
      "occ_limit_regs": null,
      "occ_limit_smem": null
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 55.830852,
      "tensor_pct": null,
      "dram_pct": 1.216341,
      "l2_pct": 24.511834,
      "warps_active_pct": 11.919707,
      "waves": null,
      "sm_active_over_elapsed": null
     }
    }
   ],
   "dominant_kernel": "void flash_fwd_kernel<Flash_fwd_kernel_traits<128, 128, 64, 4, 0, 0, bfloat16_t, Flash_kernel_traits<128, 128, 64, 4, bf",
   "primary": "unresolved_no_dominant_limiter",
   "primary_rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
   "alternative_explanation": "Mixed limiter or measurement window artifact",
   "counter_experiment": "Collect stall reasons (smsp__warp_issue_stalled_*) and l1tex/lts bytes",
   "status": "unresolved"
  },
  {
   "group": "attention:704_long:fa3:self",
   "resolution": "704x1280",
   "stage": "attention",
   "op": "flash_attn_v3",
   "shape_parameters": "{\"role\": \"self\", \"valid_keys\": 63360, \"allocated_keys\": 66880}",
   "input_sha256": "8ab8d8e91a6d5757a64db4da0bf7d6b02a4ab0fb6c1bc2ca2c2524f9a1e2f113",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/attention_hardware_extension_v1/704_long/fa3",
   "kernels": [
    {
     "kernel": "void prepare_varlen_num_blocks_kernel<1, 0>(int, int, int, const int *, const int *, const int *, const int *, const int",
     "metrics": {
      "dur_us": 2.912,
      "sm_pct": 0.00555,
      "tensor_pct": null,
      "dram_pct": 0.051476,
      "l2_pct": 0.372942,
      "warps_active_pct": 1.663289,
      "max_warps_pct": null,
      "waves": null,
      "grid": null,
      "block": null,
      "regs": null,
      "smem_dyn_kb": null,
      "smem_static_b": null,
      "l2_hit_pct": null,
      "sm_cycles_active": null,
      "cycles_elapsed": null,
      "dram_read_mb": 0.007168,
      "dram_write_mb": 0.0,
      "occ_limit_regs": null,
      "occ_limit_smem": null
     },
     "classification": "latency_or_occupancy_limited",
     "rule": "all throughputs < 40% with low achieved occupancy or < 1 wave",
     "evidence": {
      "sm_pct": 0.00555,
      "tensor_pct": null,
      "dram_pct": 0.051476,
      "l2_pct": 0.372942,
      "warps_active_pct": 1.663289,
      "waves": null,
      "sm_active_over_elapsed": null
     }
    },
    {
     "kernel": "void device_kernel<enable_sm90_or_later<FlashAttnFwdSm90<CollectiveMainloopFwdSm90<2, cute::tuple<cute::C<1>, cute::C<1>",
     "metrics": {
      "dur_us": 4680.96,
      "sm_pct": 77.71574,
      "tensor_pct": null,
      "dram_pct": 2.527247,
      "l2_pct": 26.768563,
      "warps_active_pct": 18.748163,
      "max_warps_pct": null,
      "waves": null,
      "grid": null,
      "block": null,
      "regs": null,
      "smem_dyn_kb": null,
      "smem_static_b": null,
      "l2_hit_pct": null,
      "sm_cycles_active": null,
      "cycles_elapsed": null,
      "dram_read_mb": 542.116864,
      "dram_write_mb": 27.140608,
      "occ_limit_regs": null,
      "occ_limit_smem": null
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 77.71574,
      "tensor_pct": null,
      "dram_pct": 2.527247,
      "l2_pct": 26.768563,
      "warps_active_pct": 18.748163,
      "waves": null,
      "sm_active_over_elapsed": null
     }
    }
   ],
   "dominant_kernel": "void device_kernel<enable_sm90_or_later<FlashAttnFwdSm90<CollectiveMainloopFwdSm90<2, cute::tuple<cute::C<1>, cute::C<1>",
   "primary": "compute_bound",
   "primary_rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
   "alternative_explanation": "Padding/redundant work inside nominal FLOPs; instruction mix vs tensor pipe",
   "counter_experiment": "Same dtype better tile/kernel; compare effective vs executed FLOPs",
   "status": "classified"
  },
  {
   "group": "attention:704_cross:fa2:cross",
   "resolution": "704x1280",
   "stage": "attention",
   "op": "flash_attn_v2",
   "shape_parameters": "{\"role\": \"cross\", \"valid_keys\": 512, \"allocated_keys\": 512}",
   "input_sha256": "ee5e993223e93b3cc62c90d337f72eff3586309fffbc0dc2984b73efc41077c4",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/attention_hardware_extension_v1/704_cross/fa2",
   "kernels": [
    {
     "kernel": "void flash_fwd_kernel<Flash_fwd_kernel_traits<128, 128, 64, 4, 0, 0, bfloat16_t, Flash_kernel_traits<128, 128, 64, 4, bf",
     "metrics": {
      "dur_us": 92.928,
      "sm_pct": 46.380532,
      "tensor_pct": null,
      "dram_pct": 8.888152,
      "l2_pct": 19.439977,
      "warps_active_pct": 11.87278,
      "max_warps_pct": null,
      "waves": null,
      "grid": null,
      "block": null,
      "regs": null,
      "smem_dyn_kb": null,
      "smem_static_b": null,
      "l2_hit_pct": null,
      "sm_cycles_active": null,
      "cycles_elapsed": null,
      "dram_read_mb": 29.716224,
      "dram_write_mb": 10.021376,
      "occ_limit_regs": null,
      "occ_limit_smem": null
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 46.380532,
      "tensor_pct": null,
      "dram_pct": 8.888152,
      "l2_pct": 19.439977,
      "warps_active_pct": 11.87278,
      "waves": null,
      "sm_active_over_elapsed": null
     }
    }
   ],
   "dominant_kernel": "void flash_fwd_kernel<Flash_fwd_kernel_traits<128, 128, 64, 4, 0, 0, bfloat16_t, Flash_kernel_traits<128, 128, 64, 4, bf",
   "primary": "unresolved_no_dominant_limiter",
   "primary_rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
   "alternative_explanation": "Mixed limiter or measurement window artifact",
   "counter_experiment": "Collect stall reasons (smsp__warp_issue_stalled_*) and l1tex/lts bytes",
   "status": "unresolved"
  },
  {
   "group": "attention:704_cross:fa3:cross",
   "resolution": "704x1280",
   "stage": "attention",
   "op": "flash_attn_v3",
   "shape_parameters": "{\"role\": \"cross\", \"valid_keys\": 512, \"allocated_keys\": 512}",
   "input_sha256": "ee5e993223e93b3cc62c90d337f72eff3586309fffbc0dc2984b73efc41077c4",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/attention_hardware_extension_v1/704_cross/fa3",
   "kernels": [
    {
     "kernel": "void prepare_varlen_num_blocks_kernel<1, 0>(int, int, int, const int *, const int *, const int *, const int *, const int",
     "metrics": {
      "dur_us": 2.784,
      "sm_pct": 0.005845,
      "tensor_pct": null,
      "dram_pct": 0.054224,
      "l2_pct": 0.393156,
      "warps_active_pct": 1.572167,
      "max_warps_pct": null,
      "waves": null,
      "grid": null,
      "block": null,
      "regs": null,
      "smem_dyn_kb": null,
      "smem_static_b": null,
      "l2_hit_pct": null,
      "sm_cycles_active": null,
      "cycles_elapsed": null,
      "dram_read_mb": 0.007168,
      "dram_write_mb": 0.0,
      "occ_limit_regs": null,
      "occ_limit_smem": null
     },
     "classification": "latency_or_occupancy_limited",
     "rule": "all throughputs < 40% with low achieved occupancy or < 1 wave",
     "evidence": {
      "sm_pct": 0.005845,
      "tensor_pct": null,
      "dram_pct": 0.054224,
      "l2_pct": 0.393156,
      "warps_active_pct": 1.572167,
      "waves": null,
      "sm_active_over_elapsed": null
     }
    },
    {
     "kernel": "void device_kernel<enable_sm90_or_later<FlashAttnFwdSm90<CollectiveMainloopFwdSm90<2, cute::tuple<cute::C<1>, cute::C<1>",
     "metrics": {
      "dur_us": 77.632,
      "sm_pct": 46.370683,
      "tensor_pct": null,
      "dram_pct": 10.610482,
      "l2_pct": 22.043751,
      "warps_active_pct": 18.452298,
      "max_warps_pct": null,
      "waves": null,
      "grid": null,
      "block": null,
      "regs": null,
      "smem_dyn_kb": null,
      "smem_static_b": null,
      "l2_hit_pct": null,
      "sm_cycles_active": null,
      "cycles_elapsed": null,
      "dram_read_mb": 29.719808,
      "dram_write_mb": 9.893888,
      "occ_limit_regs": null,
      "occ_limit_smem": null
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 46.370683,
      "tensor_pct": null,
      "dram_pct": 10.610482,
      "l2_pct": 22.043751,
      "warps_active_pct": 18.452298,
      "waves": null,
      "sm_active_over_elapsed": null
     }
    }
   ],
   "dominant_kernel": "void device_kernel<enable_sm90_or_later<FlashAttnFwdSm90<CollectiveMainloopFwdSm90<2, cute::tuple<cute::C<1>, cute::C<1>",
   "primary": "unresolved_no_dominant_limiter",
   "primary_rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
   "alternative_explanation": "Mixed limiter or measurement window artifact",
   "counter_experiment": "Collect stall reasons (smsp__warp_issue_stalled_*) and l1tex/lts bytes",
   "status": "unresolved"
  },
  {
   "group": "attention:480_long:fa2:self",
   "resolution": "480x832",
   "stage": "attention",
   "op": "flash_attn_v2",
   "shape_parameters": "{\"role\": \"self\", \"valid_keys\": 28080, \"allocated_keys\": 29640}",
   "input_sha256": "b972a90ab8c175e3761468ed98dc0c8304bd7ab3a3db99ebc2f9a89a7fb1b3d2",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_capability_attention_fa2_v1",
   "kernels": [
    {
     "kernel": "void flash_fwd_kernel<Flash_fwd_kernel_traits<128, 128, 64, 4, 0, 0, bfloat16_t, Flash_kernel_traits<128, 128, 64, 4, bf",
     "metrics": {
      "dur_us": 1907.808,
      "sm_pct": 55.362668,
      "tensor_pct": null,
      "dram_pct": 2.102284,
      "l2_pct": 19.837116,
      "warps_active_pct": 10.613872,
      "max_warps_pct": null,
      "waves": null,
      "grid": null,
      "block": null,
      "regs": null,
      "smem_dyn_kb": null,
      "smem_static_b": null,
      "l2_hit_pct": null,
      "sm_cycles_active": null,
      "cycles_elapsed": null,
      "dram_read_mb": 173.64096,
      "dram_write_mb": 19.356416,
      "occ_limit_regs": null,
      "occ_limit_smem": null
     },
     "classification": "unresolved_no_dominant_limiter",
     "rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
     "evidence": {
      "sm_pct": 55.362668,
      "tensor_pct": null,
      "dram_pct": 2.102284,
      "l2_pct": 19.837116,
      "warps_active_pct": 10.613872,
      "waves": null,
      "sm_active_over_elapsed": null
     }
    }
   ],
   "dominant_kernel": "void flash_fwd_kernel<Flash_fwd_kernel_traits<128, 128, 64, 4, 0, 0, bfloat16_t, Flash_kernel_traits<128, 128, 64, 4, bf",
   "primary": "unresolved_no_dominant_limiter",
   "primary_rule": "no unit >= 60% and no clear occupancy/tail signature; needs warp-stall breakdown (not collected)",
   "alternative_explanation": "Mixed limiter or measurement window artifact",
   "counter_experiment": "Collect stall reasons (smsp__warp_issue_stalled_*) and l1tex/lts bytes",
   "status": "unresolved"
  },
  {
   "group": "attention:480_long:fa3:self",
   "resolution": "480x832",
   "stage": "attention",
   "op": "flash_attn_v3",
   "shape_parameters": "{\"role\": \"self\", \"valid_keys\": 28080, \"allocated_keys\": 29640}",
   "input_sha256": "b972a90ab8c175e3761468ed98dc0c8304bd7ab3a3db99ebc2f9a89a7fb1b3d2",
   "ncu_output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_capability_attention_fa3_v1",
   "kernels": [
    {
     "kernel": "void prepare_varlen_num_blocks_kernel<1, 0>(int, int, int, const int *, const int *, const int *, const int *, const int",
     "metrics": {
      "dur_us": 2.752,
      "sm_pct": 0.005847,
      "tensor_pct": null,
      "dram_pct": 0.052288,
      "l2_pct": 0.390833,
      "warps_active_pct": 1.212442,
      "max_warps_pct": null,
      "waves": null,
      "grid": null,
      "block": null,
      "regs": null,
      "smem_dyn_kb": null,
      "smem_static_b": null,
      "l2_hit_pct": null,
      "sm_cycles_active": null,
      "cycles_elapsed": null,
      "dram_read_mb": 0.006912,
      "dram_write_mb": 0.0,
      "occ_limit_regs": null,
      "occ_limit_smem": null
     },
     "classification": "latency_or_occupancy_limited",
     "rule": "all throughputs < 40% with low achieved occupancy or < 1 wave",
     "evidence": {
      "sm_pct": 0.005847,
      "tensor_pct": null,
      "dram_pct": 0.052288,
      "l2_pct": 0.390833,
      "warps_active_pct": 1.212442,
      "waves": null,
      "sm_active_over_elapsed": null
     }
    },
    {
     "kernel": "void device_kernel<enable_sm90_or_later<FlashAttnFwdSm90<CollectiveMainloopFwdSm90<2, cute::tuple<cute::C<1>, cute::C<1>",
     "metrics": {
      "dur_us": 918.976,
      "sm_pct": 79.461747,
      "tensor_pct": null,
      "dram_pct": 4.516504,
      "l2_pct": 22.351045,
      "warps_active_pct": 18.747057,
      "max_warps_pct": null,
      "waves": null,
      "grid": null,
      "block": null,
      "regs": null,
      "smem_dyn_kb": null,
      "smem_static_b": null,
      "l2_hit_pct": null,
      "sm_cycles_active": null,
      "cycles_elapsed": null,
      "dram_read_mb": 184.964352,
      "dram_write_mb": 14.760192,
      "occ_limit_regs": null,
      "occ_limit_smem": null
     },
     "classification": "compute_bound",
     "rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
     "evidence": {
      "sm_pct": 79.461747,
      "tensor_pct": null,
      "dram_pct": 4.516504,
      "l2_pct": 22.351045,
      "warps_active_pct": 18.747057,
      "waves": null,
      "sm_active_over_elapsed": null
     }
    }
   ],
   "dominant_kernel": "void device_kernel<enable_sm90_or_later<FlashAttnFwdSm90<CollectiveMainloopFwdSm90<2, cute::tuple<cute::C<1>, cute::C<1>",
   "primary": "compute_bound",
   "primary_rule": "sm or tensor-pipe throughput >= 60% of peak sustained",
   "alternative_explanation": "Padding/redundant work inside nominal FLOPs; instruction mix vs tensor pipe",
   "counter_experiment": "Same dtype better tile/kernel; compare effective vs executed FLOPs",
   "status": "classified"
  }
 ],
 "summary": {
  "unresolved_no_dominant_limiter": 17,
  "compute_bound": 29,
  "partial_wave_tail": 13,
  "dram_bandwidth_bound": 4,
  "l2_bandwidth_bound": 6
 },
 "by_stage": {
  "dit": {
   "unresolved_no_dominant_limiter": 1,
   "compute_bound": 14,
   "partial_wave_tail": 1,
   "dram_bandwidth_bound": 4,
   "l2_bandwidth_bound": 1
  },
  "vae": {
   "partial_wave_tail": 12,
   "compute_bound": 11,
   "l2_bandwidth_bound": 5,
   "unresolved_no_dominant_limiter": 6
  },
  "output": {
   "unresolved_no_dominant_limiter": 4
  },
  "attention": {
   "unresolved_no_dominant_limiter": 6,
   "compute_bound": 4
  }
 },
 "scope": "Isolated real-input replays under NCU (kernel replay serializes and may change caching); classifications are per dominant kernel; not integrated-pipeline attribution; MFU not implied."
}
