{
  "scope": "Replay of the original ATen operator on saved real input. Evidence checks do not define a new numerical tolerance or accept a replacement implementation. Profiling durations are diagnostic, not clean timing; no NCU counters.",
  "status": "evidence_checks_passed",
  "families": [
    "dit",
    "output",
    "vae"
  ],
  "rows": [
    {
      "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/dit_rank0_cd8cbc742f5c3f37.pt",
      "sha256": "824a77bcaead8af0dacf908827027966a81c2d964d919eb4200bac0771b4a369",
      "result": {
        "op": "aten.linear.default",
        "stage": "dit",
        "shape": [
          4680,
          5120
        ],
        "exact": true,
        "max_abs": 0.0,
        "rmse": 0.0,
        "finite": true,
        "math_policy": {
          "matmul_precision": "highest",
          "matmul_allow_tf32": false,
          "cudnn_allow_tf32": true,
          "cudnn_benchmark": false,
          "cudnn_deterministic": false
        },
        "scope": "Original ATen op on saved real tensors; isolated diagnostic"
      },
      "trace": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_replay_480x832_v2/000_dit.trace.json",
      "trace_sha256": "fe3973c512ac66a158aafb4266b812669c6f053e30fb0ca53526ae0f2e164fd3",
      "kernels": [
        {
          "name": "nvjet_sm90_tst_256x144_64x4_2x1_v_bz_coopA_bias_TNT",
          "duration_us": 24.224,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 5,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 17,
            "registers per thread": 168,
            "shared memory": 221380,
            "blocks per SM": 1.0,
            "warps per SM": 12.0,
            "grid": [
              2,
              66,
              1
            ],
            "block": [
              384,
              1,
              1
            ],
            "est. achieved occupancy %": 0,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 0,
              "limitingFactors": "SMEM",
              "blockLimitRegs": 1,
              "blockLimitSharedMem": 0,
              "blockLimitWarps": 5,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 64512,
              "allocatedSharedMemPerBlock": 222464
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        }
      ],
      "checks": {
        "exit_zero": true,
        "input_hash_matches": true,
        "finite_output": true,
        "actual_gpu_kernel_recorded": true,
        "operator_recorded": true
      },
      "evidence_checks_passed": true
    },
    {
      "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/dit_rank0_9296b219534b05e0.pt",
      "sha256": "787b43c2c5ab957d3bd3c45047d646e6e407e506ea71b3249408d5d20bf00b0c",
      "result": {
        "op": "aten.linear.default",
        "stage": "dit",
        "shape": [
          1,
          4680,
          5120
        ],
        "exact": true,
        "max_abs": 0.0,
        "rmse": 0.0,
        "finite": true,
        "math_policy": {
          "matmul_precision": "highest",
          "matmul_allow_tf32": false,
          "cudnn_allow_tf32": true,
          "cudnn_benchmark": false,
          "cudnn_deterministic": false
        },
        "scope": "Original ATen op on saved real tensors; isolated diagnostic"
      },
      "trace": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_replay_480x832_v2/001_dit.trace.json",
      "trace_sha256": "b64e20d27b003ea1bf1c386df1163113bb4e5d5f18d22898e4da0b245b16d1e3",
      "kernels": [
        {
          "name": "nvjet_sm90_tst_256x144_64x4_2x1_v_bz_coopA_bias_TNT",
          "duration_us": 93.889,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 6,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 17,
            "registers per thread": 168,
            "shared memory": 221380,
            "blocks per SM": 1.0,
            "warps per SM": 12.0,
            "grid": [
              2,
              66,
              1
            ],
            "block": [
              384,
              1,
              1
            ],
            "est. achieved occupancy %": 0,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 0,
              "limitingFactors": "SMEM",
              "blockLimitRegs": 1,
              "blockLimitSharedMem": 0,
              "blockLimitWarps": 5,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 64512,
              "allocatedSharedMemPerBlock": 222464
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        }
      ],
      "checks": {
        "exit_zero": true,
        "input_hash_matches": true,
        "finite_output": true,
        "actual_gpu_kernel_recorded": true,
        "operator_recorded": true
      },
      "evidence_checks_passed": true
    },
    {
      "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/dit_rank0_f56603026672560c.pt",
      "sha256": "4a5b492ee606c1a2c9fe4844a0c3125609fa599f4c027f56b4329c89d2879b7e",
      "result": {
        "op": "aten.linear.default",
        "stage": "dit",
        "shape": [
          1,
          4680,
          5120
        ],
        "exact": true,
        "max_abs": 0.0,
        "rmse": 0.0,
        "finite": true,
        "math_policy": {
          "matmul_precision": "highest",
          "matmul_allow_tf32": false,
          "cudnn_allow_tf32": true,
          "cudnn_benchmark": false,
          "cudnn_deterministic": false
        },
        "scope": "Original ATen op on saved real tensors; isolated diagnostic"
      },
      "trace": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_replay_480x832_v2/002_dit.trace.json",
      "trace_sha256": "33c1e7526180abeac91cc75ab706402cf93a70d9aba36250b5ceb127110e3925",
      "kernels": [
        {
          "name": "nvjet_sm90_tst_256x144_64x4_2x1_v_bz_coopA_bias_TNT",
          "duration_us": 290.945,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 6,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 17,
            "registers per thread": 168,
            "shared memory": 221380,
            "blocks per SM": 1.0,
            "warps per SM": 12.0,
            "grid": [
              2,
              66,
              1
            ],
            "block": [
              384,
              1,
              1
            ],
            "est. achieved occupancy %": 0,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 0,
              "limitingFactors": "SMEM",
              "blockLimitRegs": 1,
              "blockLimitSharedMem": 0,
              "blockLimitWarps": 5,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 64512,
              "allocatedSharedMemPerBlock": 222464
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        }
      ],
      "checks": {
        "exit_zero": true,
        "input_hash_matches": true,
        "finite_output": true,
        "actual_gpu_kernel_recorded": true,
        "operator_recorded": true
      },
      "evidence_checks_passed": true
    },
    {
      "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/dit_rank0_10a11d3772070763.pt",
      "sha256": "ea1184eaad12555ac80bf5ee33de7aa2251926b289572cadadb4043884848587",
      "result": {
        "op": "aten.linear.default",
        "stage": "dit",
        "shape": [
          3,
          5120
        ],
        "exact": true,
        "max_abs": 0.0,
        "rmse": 0.0,
        "finite": true,
        "math_policy": {
          "matmul_precision": "highest",
          "matmul_allow_tf32": false,
          "cudnn_allow_tf32": true,
          "cudnn_benchmark": false,
          "cudnn_deterministic": false
        },
        "scope": "Original ATen op on saved real tensors; isolated diagnostic"
      },
      "trace": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_replay_480x832_v2/003_dit.trace.json",
      "trace_sha256": "55367134f6d06de731e86b33c8495bee5200257d2ad15548a9867cca92d6cf5e",
      "kernels": [
        {
          "name": "nvjet_sm90_tst_64x8_64x16_4x1_v_bz_bias_TNT",
          "duration_us": 2.88,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 5,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 16,
            "registers per thread": 168,
            "shared memory": 164308,
            "blocks per SM": 0.606061,
            "warps per SM": 7.272727,
            "grid": [
              4,
              20,
              1
            ],
            "block": [
              384,
              1,
              1
            ],
            "est. achieved occupancy %": 0,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 0,
              "limitingFactors": "SMEM",
              "blockLimitRegs": 1,
              "blockLimitSharedMem": 0,
              "blockLimitWarps": 5,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 64512,
              "allocatedSharedMemPerBlock": 165376
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        }
      ],
      "checks": {
        "exit_zero": true,
        "input_hash_matches": true,
        "finite_output": true,
        "actual_gpu_kernel_recorded": true,
        "operator_recorded": true
      },
      "evidence_checks_passed": true
    },
    {
      "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/dit_rank0_ca590b3a668a9ecf.pt",
      "sha256": "a6cb8d5375b991c9c5d1a1e5674fa86eb61d10648b2ccf71bb7ec40c56f2ac1e",
      "result": {
        "op": "aten.linear.default",
        "stage": "dit",
        "shape": [
          3,
          5120
        ],
        "exact": true,
        "max_abs": 0.0,
        "rmse": 0.0,
        "finite": true,
        "math_policy": {
          "matmul_precision": "highest",
          "matmul_allow_tf32": false,
          "cudnn_allow_tf32": true,
          "cudnn_benchmark": false,
          "cudnn_deterministic": false
        },
        "scope": "Original ATen op on saved real tensors; isolated diagnostic"
      },
      "trace": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_replay_480x832_v2/004_dit.trace.json",
      "trace_sha256": "1fc44b5f196e7bd835b32c4b648b0085fbe0b7df2512d798771bc33ddbb425fb",
      "kernels": [
        {
          "name": "nvjet_sm90_tst_64x8_64x16_4x1_v_bz_bias_TNT",
          "duration_us": 17.023,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 5,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 16,
            "registers per thread": 168,
            "shared memory": 164308,
            "blocks per SM": 0.606061,
            "warps per SM": 7.272727,
            "grid": [
              4,
              20,
              1
            ],
            "block": [
              384,
              1,
              1
            ],
            "est. achieved occupancy %": 0,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 0,
              "limitingFactors": "SMEM",
              "blockLimitRegs": 1,
              "blockLimitSharedMem": 0,
              "blockLimitWarps": 5,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 64512,
              "allocatedSharedMemPerBlock": 165376
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        }
      ],
      "checks": {
        "exit_zero": true,
        "input_hash_matches": true,
        "finite_output": true,
        "actual_gpu_kernel_recorded": true,
        "operator_recorded": true
      },
      "evidence_checks_passed": true
    },
    {
      "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/dit_rank0_22805ed4eb786e66.pt",
      "sha256": "d6a2e382ecf0fdad711df2cd1bcb239a004321ae666d26aa04920718cf49bf81",
      "result": {
        "op": "aten.linear.default",
        "stage": "dit",
        "shape": [
          1,
          3,
          30720
        ],
        "exact": true,
        "max_abs": 0.0,
        "rmse": 0.0,
        "finite": true,
        "math_policy": {
          "matmul_precision": "highest",
          "matmul_allow_tf32": false,
          "cudnn_allow_tf32": true,
          "cudnn_benchmark": false,
          "cudnn_deterministic": false
        },
        "scope": "Original ATen op on saved real tensors; isolated diagnostic"
      },
      "trace": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_replay_480x832_v2/005_dit.trace.json",
      "trace_sha256": "1915c84b146a56a0dbb97bbbfae1eee5a8a834c2b091173de810d5b3b2b37abd",
      "kernels": [
        {
          "name": "nvjet_sm90_tst_64x8_64x16_2x1_v_bz_bias_TNT",
          "duration_us": 78.528,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 6,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 17,
            "registers per thread": 168,
            "shared memory": 164308,
            "blocks per SM": 1.0,
            "warps per SM": 12.0,
            "grid": [
              2,
              66,
              1
            ],
            "block": [
              384,
              1,
              1
            ],
            "est. achieved occupancy %": 0,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 0,
              "limitingFactors": "SMEM",
              "blockLimitRegs": 1,
              "blockLimitSharedMem": 0,
              "blockLimitWarps": 5,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 64512,
              "allocatedSharedMemPerBlock": 165376
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        }
      ],
      "checks": {
        "exit_zero": true,
        "input_hash_matches": true,
        "finite_output": true,
        "actual_gpu_kernel_recorded": true,
        "operator_recorded": true
      },
      "evidence_checks_passed": true
    },
    {
      "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/dit_rank0_fcebe966800b94b1.pt",
      "sha256": "4f1e9988bf86b3f91f274f9b8c3bf33bf42b39945181ce6f09fbce8e0132cb6f",
      "result": {
        "op": "aten.linear.default",
        "stage": "dit",
        "shape": [
          1,
          1170,
          15360
        ],
        "exact": true,
        "max_abs": 0.0,
        "rmse": 0.0,
        "finite": true,
        "math_policy": {
          "matmul_precision": "highest",
          "matmul_allow_tf32": false,
          "cudnn_allow_tf32": true,
          "cudnn_benchmark": false,
          "cudnn_deterministic": false
        },
        "scope": "Original ATen op on saved real tensors; isolated diagnostic"
      },
      "trace": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_replay_480x832_v2/006_dit.trace.json",
      "trace_sha256": "b88096a8e6eb0e66e6acb7fd33c1aa33e120e33da8b6cbfe5820273520d35f22",
      "kernels": [
        {
          "name": "nvjet_sm90_tst_256x152_64x4_1x2_h_bz_coopA_bias_TNT",
          "duration_us": 251.458,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 6,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 17,
            "registers per thread": 168,
            "shared memory": 225476,
            "blocks per SM": 1.0,
            "warps per SM": 12.0,
            "grid": [
              2,
              66,
              1
            ],
            "block": [
              384,
              1,
              1
            ],
            "est. achieved occupancy %": 0,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 0,
              "limitingFactors": "SMEM",
              "blockLimitRegs": 1,
              "blockLimitSharedMem": 0,
              "blockLimitWarps": 5,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 64512,
              "allocatedSharedMemPerBlock": 226560
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        }
      ],
      "checks": {
        "exit_zero": true,
        "input_hash_matches": true,
        "finite_output": true,
        "actual_gpu_kernel_recorded": true,
        "operator_recorded": true
      },
      "evidence_checks_passed": true
    },
    {
      "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/dit_rank0_10a9e2fea034c6cf.pt",
      "sha256": "a64f7da6eb1fcc9779d40258d974eb529c2397aaba4df6e32363d7c5d4e73065",
      "result": {
        "op": "aten.linear.default",
        "stage": "dit",
        "shape": [
          1,
          1170,
          5120
        ],
        "exact": true,
        "max_abs": 0.0,
        "rmse": 0.0,
        "finite": true,
        "math_policy": {
          "matmul_precision": "highest",
          "matmul_allow_tf32": false,
          "cudnn_allow_tf32": true,
          "cudnn_benchmark": false,
          "cudnn_deterministic": false
        },
        "scope": "Original ATen op on saved real tensors; isolated diagnostic"
      },
      "trace": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_replay_480x832_v2/007_dit.trace.json",
      "trace_sha256": "8f6dfc221248878721be9741abffc9ca69947ebbba496c1d7849986187a7c93f",
      "kernels": [
        {
          "name": "nvjet_sm90_tst_128x200_64x5_2x1_v_bz_coopA_bias_TNT",
          "duration_us": 85.281,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 6,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 16,
            "registers per thread": 168,
            "shared memory": 226492,
            "blocks per SM": 1.0,
            "warps per SM": 12.0,
            "grid": [
              2,
              66,
              1
            ],
            "block": [
              384,
              1,
              1
            ],
            "est. achieved occupancy %": 0,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 0,
              "limitingFactors": "SMEM",
              "blockLimitRegs": 1,
              "blockLimitSharedMem": 0,
              "blockLimitWarps": 5,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 64512,
              "allocatedSharedMemPerBlock": 227584
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        }
      ],
      "checks": {
        "exit_zero": true,
        "input_hash_matches": true,
        "finite_output": true,
        "actual_gpu_kernel_recorded": true,
        "operator_recorded": true
      },
      "evidence_checks_passed": true
    },
    {
      "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_6638f430ac58a438.pt",
      "sha256": "32bcd9b4d8be26c876626649d4ff2e623424efe309222bcd7e80afc47310ded4",
      "result": {
        "op": "aten.conv3d.default",
        "stage": "vae",
        "shape": [
          1,
          16,
          3,
          60,
          104
        ],
        "exact": true,
        "max_abs": 0.0,
        "rmse": 0.0,
        "finite": true,
        "math_policy": {
          "matmul_precision": "highest",
          "matmul_allow_tf32": false,
          "cudnn_allow_tf32": true,
          "cudnn_benchmark": false,
          "cudnn_deterministic": false
        },
        "scope": "Original ATen op on saved real tensors; isolated diagnostic"
      },
      "trace": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_replay_480x832_v2/008_vae.trace.json",
      "trace_sha256": "ffd9abb9e1212d20346f78e9646cc77515899d5a81f7ca53e3d2b3060f5f14f6",
      "kernels": [
        {
          "name": "void cutlass::Kernel2<cutlass_80_tensorop_s1688gemm_128x64_16x6_nn_align4>(cutlass_80_tensorop_s1688gemm_128x64_16x6_nn_align4::Params)",
          "duration_us": 3.872,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 4,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 17,
            "registers per thread": 136,
            "shared memory": 73728,
            "blocks per SM": 1.151515,
            "warps per SM": 4.606061,
            "grid": [
              8,
              19,
              1
            ],
            "block": [
              128,
              1,
              1
            ],
            "est. achieved occupancy %": 0,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 0,
              "limitingFactors": "SMEM",
              "blockLimitRegs": 3,
              "blockLimitSharedMem": 0,
              "blockLimitWarps": 16,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 17408,
              "allocatedSharedMemPerBlock": 74752
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        },
        {
          "name": "void at::native::elementwise_kernel<128, 2, at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<float> >(at::TensorIteratorBase&, at::native::CUDAFunctor_add<float> const&)::{lambda(int)#1}>(int, at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<float> >(at::TensorIteratorBase&, at::native::CUDAFunctor_add<float> const&)::{lambda(int)#1})",
          "duration_us": 2.336,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 7,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 25,
            "registers per thread": 18,
            "shared memory": 0,
            "blocks per SM": 8.863636,
            "warps per SM": 35.454544,
            "grid": [
              1170,
              1,
              1
            ],
            "block": [
              128,
              1,
              1
            ],
            "est. achieved occupancy %": 55,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 16,
              "limitingFactors": "WARPS",
              "blockLimitRegs": 21,
              "blockLimitSharedMem": 228,
              "blockLimitWarps": 16,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 3072,
              "allocatedSharedMemPerBlock": 1024
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        }
      ],
      "checks": {
        "exit_zero": true,
        "input_hash_matches": true,
        "finite_output": true,
        "actual_gpu_kernel_recorded": true,
        "operator_recorded": true
      },
      "evidence_checks_passed": true
    },
    {
      "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_29ef146661136bf4.pt",
      "sha256": "2b9e80cee384305d79045df290612b04bcfe041bd67dc69a2bc096e45a6c5872",
      "result": {
        "op": "aten.conv3d.default",
        "stage": "vae",
        "shape": [
          1,
          384,
          1,
          60,
          26
        ],
        "exact": true,
        "max_abs": 0.0,
        "rmse": 0.0,
        "finite": true,
        "math_policy": {
          "matmul_precision": "highest",
          "matmul_allow_tf32": false,
          "cudnn_allow_tf32": true,
          "cudnn_benchmark": false,
          "cudnn_deterministic": false
        },
        "scope": "Original ATen op on saved real tensors; isolated diagnostic"
      },
      "trace": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_replay_480x832_v2/009_vae.trace.json",
      "trace_sha256": "2ed2af2526ac2a519a16747285f649426b641456ec56a042fb2087147ca09f18",
      "kernels": [
        {
          "name": "void cudnn::engines_precompiled::nchwToNhwcKernel<float, float, float, false, true, (cudnnKernelDataType_t)2>(cudnn::engines_precompiled::nchw2nhwc_params_t<float>, float const*, float*)",
          "duration_us": 1.696,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 4,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 12,
            "registers per thread": 32,
            "shared memory": 4224,
            "blocks per SM": 1.234848,
            "warps per SM": 9.878788,
            "grid": [
              163,
              1,
              1
            ],
            "block": [
              256,
              1,
              1
            ],
            "est. achieved occupancy %": 15,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 8,
              "limitingFactors": "WARPS|REGS",
              "blockLimitRegs": 8,
              "blockLimitSharedMem": 44,
              "blockLimitWarps": 8,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 8192,
              "allocatedSharedMemPerBlock": 5248
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        },
        {
          "name": "void cudnn::engines_precompiled::nchwToNhwcKernel<float, float, float, false, true, (cudnnKernelDataType_t)2>(cudnn::engines_precompiled::nchw2nhwc_params_t<float>, float const*, float*)",
          "duration_us": 1.824,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 4,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 16,
            "registers per thread": 32,
            "shared memory": 4224,
            "blocks per SM": 2.909091,
            "warps per SM": 23.272728,
            "grid": [
              1,
              1,
              384
            ],
            "block": [
              256,
              1,
              1
            ],
            "est. achieved occupancy %": 36,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 8,
              "limitingFactors": "WARPS|REGS",
              "blockLimitRegs": 8,
              "blockLimitSharedMem": 44,
              "blockLimitWarps": 8,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 8192,
              "allocatedSharedMemPerBlock": 5248
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        },
        {
          "name": "sm80_xmma_fprop_implicit_gemm_tf32f32_tf32f32_f32_nhwckrsc_nchw_tilesize128x64x32_stage5_warpsize2x2x1_g1_tensor16x8x8_execute_kernel__5x_cudnn",
          "duration_us": 15.457,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 4,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 21,
            "registers per thread": 168,
            "shared memory": 122880,
            "blocks per SM": 0.590909,
            "warps per SM": 2.363636,
            "grid": [
              6,
              13,
              1
            ],
            "block": [
              128,
              1,
              1
            ],
            "est. achieved occupancy %": 0,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 0,
              "limitingFactors": "SMEM",
              "blockLimitRegs": 3,
              "blockLimitSharedMem": 0,
              "blockLimitWarps": 16,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 21504,
              "allocatedSharedMemPerBlock": 123904
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        },
        {
          "name": "void at::native::elementwise_kernel<128, 2, at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<float> >(at::TensorIteratorBase&, at::native::CUDAFunctor_add<float> const&)::{lambda(int)#1}>(int, at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<float> >(at::TensorIteratorBase&, at::native::CUDAFunctor_add<float> const&)::{lambda(int)#1})",
          "duration_us": 3.104,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 7,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 31,
            "registers per thread": 18,
            "shared memory": 0,
            "blocks per SM": 17.727272,
            "warps per SM": 70.909088,
            "grid": [
              2340,
              1,
              1
            ],
            "block": [
              128,
              1,
              1
            ],
            "est. achieved occupancy %": 100,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 16,
              "limitingFactors": "WARPS",
              "blockLimitRegs": 21,
              "blockLimitSharedMem": 228,
              "blockLimitWarps": 16,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 3072,
              "allocatedSharedMemPerBlock": 1024
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        }
      ],
      "checks": {
        "exit_zero": true,
        "input_hash_matches": true,
        "finite_output": true,
        "actual_gpu_kernel_recorded": true,
        "operator_recorded": true
      },
      "evidence_checks_passed": true
    },
    {
      "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_551fa1ff3095f40c.pt",
      "sha256": "6309c12f24013edabd88edced7c2ce22672322d203206b12ba218c88696effb5",
      "result": {
        "op": "aten.conv3d.default",
        "stage": "vae",
        "shape": [
          1,
          384,
          1,
          60,
          26
        ],
        "exact": true,
        "max_abs": 0.0,
        "rmse": 0.0,
        "finite": true,
        "math_policy": {
          "matmul_precision": "highest",
          "matmul_allow_tf32": false,
          "cudnn_allow_tf32": true,
          "cudnn_benchmark": false,
          "cudnn_deterministic": false
        },
        "scope": "Original ATen op on saved real tensors; isolated diagnostic"
      },
      "trace": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_replay_480x832_v2/010_vae.trace.json",
      "trace_sha256": "19edee2ec48a654f70c5630ce3b4db42f221d3a4de6c29016fa97ec92b487379",
      "kernels": [
        {
          "name": "void cudnn::engines_precompiled::nchwToNhwcKernel<float, float, float, false, true, (cudnnKernelDataType_t)0>(cudnn::engines_precompiled::nchw2nhwc_params_t<float>, float const*, float*)",
          "duration_us": 5.824,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 4,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 12,
            "registers per thread": 32,
            "shared memory": 4224,
            "blocks per SM": 14.818182,
            "warps per SM": 118.545456,
            "grid": [
              163,
              12,
              1
            ],
            "block": [
              256,
              1,
              1
            ],
            "est. achieved occupancy %": 100,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 8,
              "limitingFactors": "WARPS|REGS",
              "blockLimitRegs": 8,
              "blockLimitSharedMem": 44,
              "blockLimitWarps": 8,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 8192,
              "allocatedSharedMemPerBlock": 5248
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        },
        {
          "name": "void cudnn::engines_precompiled::nchwToNhwcKernel<float, float, float, false, true, (cudnnKernelDataType_t)0>(cudnn::engines_precompiled::nchw2nhwc_params_t<float>, float const*, float*)",
          "duration_us": 11.296,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 4,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 16,
            "registers per thread": 32,
            "shared memory": 4224,
            "blocks per SM": 34.909092,
            "warps per SM": 279.272736,
            "grid": [
              1,
              12,
              384
            ],
            "block": [
              256,
              1,
              1
            ],
            "est. achieved occupancy %": 100,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 8,
              "limitingFactors": "WARPS|REGS",
              "blockLimitRegs": 8,
              "blockLimitSharedMem": 44,
              "blockLimitWarps": 8,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 8192,
              "allocatedSharedMemPerBlock": 5248
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        },
        {
          "name": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize64x64x32_warpgroupsize1x1x1_g1_execute_segment_k_off_kernel__5x_cudnn",
          "duration_us": 92.256,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 4,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 22,
            "registers per thread": 168,
            "shared memory": 231424,
            "blocks per SM": 0.909091,
            "warps per SM": 10.909091,
            "grid": [
              60,
              2,
              1
            ],
            "block": [
              384,
              1,
              1
            ],
            "est. achieved occupancy %": 0,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 0,
              "limitingFactors": "SMEM",
              "blockLimitRegs": 1,
              "blockLimitSharedMem": 0,
              "blockLimitWarps": 5,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 64512,
              "allocatedSharedMemPerBlock": 232448
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        },
        {
          "name": "void cudnn::engines_precompiled::nhwcToNchwKernel<float, float, float, true, false, (cudnnKernelDataType_t)0>(cudnn::engines_precompiled::nhwc2nchw_params_t<float>, float const*, float*)",
          "duration_us": 2.368,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 4,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 26,
            "registers per thread": 32,
            "shared memory": 4224,
            "blocks per SM": 4.454545,
            "warps per SM": 35.636364,
            "grid": [
              49,
              12,
              1
            ],
            "block": [
              256,
              1,
              1
            ],
            "est. achieved occupancy %": 56,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 8,
              "limitingFactors": "WARPS|REGS",
              "blockLimitRegs": 8,
              "blockLimitSharedMem": 44,
              "blockLimitWarps": 8,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 8192,
              "allocatedSharedMemPerBlock": 5248
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        },
        {
          "name": "void at::native::elementwise_kernel<128, 2, at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<float> >(at::TensorIteratorBase&, at::native::CUDAFunctor_add<float> const&)::{lambda(int)#1}>(int, at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<float> >(at::TensorIteratorBase&, at::native::CUDAFunctor_add<float> const&)::{lambda(int)#1})",
          "duration_us": 3.232,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 7,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 36,
            "registers per thread": 18,
            "shared memory": 0,
            "blocks per SM": 17.727272,
            "warps per SM": 70.909088,
            "grid": [
              2340,
              1,
              1
            ],
            "block": [
              128,
              1,
              1
            ],
            "est. achieved occupancy %": 100,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 16,
              "limitingFactors": "WARPS",
              "blockLimitRegs": 21,
              "blockLimitSharedMem": 228,
              "blockLimitWarps": 16,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 3072,
              "allocatedSharedMemPerBlock": 1024
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        }
      ],
      "checks": {
        "exit_zero": true,
        "input_hash_matches": true,
        "finite_output": true,
        "actual_gpu_kernel_recorded": true,
        "operator_recorded": true
      },
      "evidence_checks_passed": true
    },
    {
      "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_c16d11a7624e1a8c.pt",
      "sha256": "1e43418295ca6422f27afcf0b9940aa818fbb12fcf1013c12dfcf7a60f86e7d7",
      "result": {
        "op": "aten.conv3d.default",
        "stage": "vae",
        "shape": [
          1,
          384,
          1,
          120,
          52
        ],
        "exact": true,
        "max_abs": 0.0,
        "rmse": 0.0,
        "finite": true,
        "math_policy": {
          "matmul_precision": "highest",
          "matmul_allow_tf32": false,
          "cudnn_allow_tf32": true,
          "cudnn_benchmark": false,
          "cudnn_deterministic": false
        },
        "scope": "Original ATen op on saved real tensors; isolated diagnostic"
      },
      "trace": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_replay_480x832_v2/011_vae.trace.json",
      "trace_sha256": "d219fed997a5569d3377d5a4d27ccc850395b8af1c1d7e5878ba90a1c31dbf49",
      "kernels": [
        {
          "name": "void cutlass::Kernel2<cutlass_80_tensorop_s1688gemm_64x128_32x3_nn_align4>(cutlass_80_tensorop_s1688gemm_64x128_32x3_nn_align4::Params)",
          "duration_us": 11.008,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 4,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 18,
            "registers per thread": 168,
            "shared memory": 73728,
            "blocks per SM": 2.363636,
            "warps per SM": 9.454545,
            "grid": [
              24,
              13,
              1
            ],
            "block": [
              128,
              1,
              1
            ],
            "est. achieved occupancy %": 0,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 0,
              "limitingFactors": "SMEM",
              "blockLimitRegs": 3,
              "blockLimitSharedMem": 0,
              "blockLimitWarps": 16,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 21504,
              "allocatedSharedMemPerBlock": 74752
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        },
        {
          "name": "void at::native::elementwise_kernel<128, 2, at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<float> >(at::TensorIteratorBase&, at::native::CUDAFunctor_add<float> const&)::{lambda(int)#1}>(int, at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<float> >(at::TensorIteratorBase&, at::native::CUDAFunctor_add<float> const&)::{lambda(int)#1})",
          "duration_us": 7.392,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 7,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 26,
            "registers per thread": 18,
            "shared memory": 0,
            "blocks per SM": 70.909088,
            "warps per SM": 283.636353,
            "grid": [
              9360,
              1,
              1
            ],
            "block": [
              128,
              1,
              1
            ],
            "est. achieved occupancy %": 100,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 16,
              "limitingFactors": "WARPS",
              "blockLimitRegs": 21,
              "blockLimitSharedMem": 228,
              "blockLimitWarps": 16,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 3072,
              "allocatedSharedMemPerBlock": 1024
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        }
      ],
      "checks": {
        "exit_zero": true,
        "input_hash_matches": true,
        "finite_output": true,
        "actual_gpu_kernel_recorded": true,
        "operator_recorded": true
      },
      "evidence_checks_passed": true
    },
    {
      "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_081e405bd1e07dce.pt",
      "sha256": "3193a32ddc024e9d34eb5446f9909408dde38d963ebc5cb9654998ecf661a820",
      "result": {
        "op": "aten.conv3d.default",
        "stage": "vae",
        "shape": [
          1,
          384,
          1,
          120,
          52
        ],
        "exact": true,
        "max_abs": 0.0,
        "rmse": 0.0,
        "finite": true,
        "math_policy": {
          "matmul_precision": "highest",
          "matmul_allow_tf32": false,
          "cudnn_allow_tf32": true,
          "cudnn_benchmark": false,
          "cudnn_deterministic": false
        },
        "scope": "Original ATen op on saved real tensors; isolated diagnostic"
      },
      "trace": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_replay_480x832_v2/012_vae.trace.json",
      "trace_sha256": "3bed073d85a198cc0b7060c5b3e8574ac9482bd08fad1467777e7086a9e1979a",
      "kernels": [
        {
          "name": "void cudnn::engines_precompiled::nchwToNhwcKernel<float, float, float, false, true, (cudnnKernelDataType_t)0>(cudnn::engines_precompiled::nchw2nhwc_params_t<float>, float const*, float*)",
          "duration_us": 8.832,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 4,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 12,
            "registers per thread": 32,
            "shared memory": 4224,
            "blocks per SM": 28.09091,
            "warps per SM": 224.72728,
            "grid": [
              618,
              6,
              1
            ],
            "block": [
              256,
              1,
              1
            ],
            "est. achieved occupancy %": 100,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 8,
              "limitingFactors": "WARPS|REGS",
              "blockLimitRegs": 8,
              "blockLimitSharedMem": 44,
              "blockLimitWarps": 8,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 8192,
              "allocatedSharedMemPerBlock": 5248
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        },
        {
          "name": "void cudnn::engines_precompiled::nchwToNhwcKernel<float, float, float, false, true, (cudnnKernelDataType_t)0>(cudnn::engines_precompiled::nchw2nhwc_params_t<float>, float const*, float*)",
          "duration_us": 6.048,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 4,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 16,
            "registers per thread": 32,
            "shared memory": 4224,
            "blocks per SM": 17.454546,
            "warps per SM": 139.636368,
            "grid": [
              1,
              6,
              384
            ],
            "block": [
              256,
              1,
              1
            ],
            "est. achieved occupancy %": 100,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 8,
              "limitingFactors": "WARPS|REGS",
              "blockLimitRegs": 8,
              "blockLimitSharedMem": 44,
              "blockLimitWarps": 8,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 8192,
              "allocatedSharedMemPerBlock": 5248
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        },
        {
          "name": "void cask_plugin__5x_cudnn::xmma__5x_cudnn::init_device_workspace_kernel<xmma__5x_cudnn::implicit_gemm::fprop::Warp_specialized_params_non_template<xmma__5x_cudnn::Grid_constant_params> >(xmma__5x_cudnn::implicit_gemm::fprop::Warp_specialized_params_non_template<xmma__5x_cudnn::Grid_constant_params>, bool)",
          "duration_us": 2.176,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 4,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 20,
            "registers per thread": 16,
            "shared memory": 0,
            "blocks per SM": 0.007576,
            "warps per SM": 0.000237,
            "grid": [
              1,
              1,
              1
            ],
            "block": [
              1,
              1,
              1
            ],
            "est. achieved occupancy %": 0,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 32,
              "limitingFactors": "BLOCKS",
              "blockLimitRegs": 128,
              "blockLimitSharedMem": 228,
              "blockLimitWarps": 64,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 512,
              "allocatedSharedMemPerBlock": 1024
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        },
        {
          "name": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_k_on_kernel__5x_cudnn",
          "duration_us": 119.968,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 4,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 23,
            "registers per thread": 168,
            "shared memory": 231424,
            "blocks per SM": 1.0,
            "warps per SM": 12.0,
            "grid": [
              132,
              1,
              1
            ],
            "block": [
              384,
              1,
              1
            ],
            "est. achieved occupancy %": 0,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 0,
              "limitingFactors": "SMEM",
              "blockLimitRegs": 1,
              "blockLimitSharedMem": 0,
              "blockLimitWarps": 5,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 64512,
              "allocatedSharedMemPerBlock": 232448
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        },
        {
          "name": "void cudnn::engines_precompiled::nhwcToNchwKernel<float, float, float, true, false, (cudnnKernelDataType_t)0>(cudnn::engines_precompiled::nhwc2nchw_params_t<float>, float const*, float*)",
          "duration_us": 5.312,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 4,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 27,
            "registers per thread": 32,
            "shared memory": 4224,
            "blocks per SM": 17.727272,
            "warps per SM": 141.818176,
            "grid": [
              195,
              12,
              1
            ],
            "block": [
              256,
              1,
              1
            ],
            "est. achieved occupancy %": 100,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 8,
              "limitingFactors": "WARPS|REGS",
              "blockLimitRegs": 8,
              "blockLimitSharedMem": 44,
              "blockLimitWarps": 8,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 8192,
              "allocatedSharedMemPerBlock": 5248
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        },
        {
          "name": "void at::native::elementwise_kernel<128, 2, at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<float> >(at::TensorIteratorBase&, at::native::CUDAFunctor_add<float> const&)::{lambda(int)#1}>(int, at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<float> >(at::TensorIteratorBase&, at::native::CUDAFunctor_add<float> const&)::{lambda(int)#1})",
          "duration_us": 7.552,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 7,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 37,
            "registers per thread": 18,
            "shared memory": 0,
            "blocks per SM": 70.909088,
            "warps per SM": 283.636353,
            "grid": [
              9360,
              1,
              1
            ],
            "block": [
              128,
              1,
              1
            ],
            "est. achieved occupancy %": 100,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 16,
              "limitingFactors": "WARPS",
              "blockLimitRegs": 21,
              "blockLimitSharedMem": 228,
              "blockLimitWarps": 16,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 3072,
              "allocatedSharedMemPerBlock": 1024
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        }
      ],
      "checks": {
        "exit_zero": true,
        "input_hash_matches": true,
        "finite_output": true,
        "actual_gpu_kernel_recorded": true,
        "operator_recorded": true
      },
      "evidence_checks_passed": true
    },
    {
      "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_8601bc3a88374d99.pt",
      "sha256": "c7f10fe8b8ca84229e785645c70c003456c3950c79cdd176694e5bc86c9247f4",
      "result": {
        "op": "aten.conv3d.default",
        "stage": "vae",
        "shape": [
          1,
          384,
          1,
          120,
          52
        ],
        "exact": true,
        "max_abs": 0.0,
        "rmse": 0.0,
        "finite": true,
        "math_policy": {
          "matmul_precision": "highest",
          "matmul_allow_tf32": false,
          "cudnn_allow_tf32": true,
          "cudnn_benchmark": false,
          "cudnn_deterministic": false
        },
        "scope": "Original ATen op on saved real tensors; isolated diagnostic"
      },
      "trace": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_replay_480x832_v2/013_vae.trace.json",
      "trace_sha256": "83ea7aa0395a48a682f41d310f6e1f134334b98540b6660943aa0a0b7181c56e",
      "kernels": [
        {
          "name": "void cudnn::engines_precompiled::nchwToNhwcKernel<float, float, float, false, true, (cudnnKernelDataType_t)0>(cudnn::engines_precompiled::nchw2nhwc_params_t<float>, float const*, float*)",
          "duration_us": 18.784,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 4,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 12,
            "registers per thread": 32,
            "shared memory": 4224,
            "blocks per SM": 56.18182,
            "warps per SM": 449.454559,
            "grid": [
              618,
              12,
              1
            ],
            "block": [
              256,
              1,
              1
            ],
            "est. achieved occupancy %": 100,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 8,
              "limitingFactors": "WARPS|REGS",
              "blockLimitRegs": 8,
              "blockLimitSharedMem": 44,
              "blockLimitWarps": 8,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 8192,
              "allocatedSharedMemPerBlock": 5248
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        },
        {
          "name": "void cudnn::engines_precompiled::nchwToNhwcKernel<float, float, float, false, true, (cudnnKernelDataType_t)0>(cudnn::engines_precompiled::nchw2nhwc_params_t<float>, float const*, float*)",
          "duration_us": 11.52,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 4,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 16,
            "registers per thread": 32,
            "shared memory": 4224,
            "blocks per SM": 34.909092,
            "warps per SM": 279.272736,
            "grid": [
              1,
              12,
              384
            ],
            "block": [
              256,
              1,
              1
            ],
            "est. achieved occupancy %": 100,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 8,
              "limitingFactors": "WARPS|REGS",
              "blockLimitRegs": 8,
              "blockLimitSharedMem": 44,
              "blockLimitWarps": 8,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 8192,
              "allocatedSharedMemPerBlock": 5248
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        },
        {
          "name": "void cask_plugin__5x_cudnn::xmma__5x_cudnn::init_device_workspace_kernel<xmma__5x_cudnn::implicit_gemm::fprop::Warp_specialized_params_non_template<xmma__5x_cudnn::Grid_constant_params> >(xmma__5x_cudnn::implicit_gemm::fprop::Warp_specialized_params_non_template<xmma__5x_cudnn::Grid_constant_params>, bool)",
          "duration_us": 2.24,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 4,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 20,
            "registers per thread": 16,
            "shared memory": 0,
            "blocks per SM": 0.007576,
            "warps per SM": 0.000237,
            "grid": [
              1,
              1,
              1
            ],
            "block": [
              1,
              1,
              1
            ],
            "est. achieved occupancy %": 0,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 32,
              "limitingFactors": "BLOCKS",
              "blockLimitRegs": 128,
              "blockLimitSharedMem": 228,
              "blockLimitWarps": 64,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 512,
              "allocatedSharedMemPerBlock": 1024
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        },
        {
          "name": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_k_on_kernel__5x_cudnn",
          "duration_us": 234.433,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 4,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 23,
            "registers per thread": 168,
            "shared memory": 231424,
            "blocks per SM": 1.0,
            "warps per SM": 12.0,
            "grid": [
              132,
              1,
              1
            ],
            "block": [
              384,
              1,
              1
            ],
            "est. achieved occupancy %": 0,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 0,
              "limitingFactors": "SMEM",
              "blockLimitRegs": 1,
              "blockLimitSharedMem": 0,
              "blockLimitWarps": 5,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 64512,
              "allocatedSharedMemPerBlock": 232448
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        },
        {
          "name": "void cudnn::engines_precompiled::nhwcToNchwKernel<float, float, float, true, false, (cudnnKernelDataType_t)0>(cudnn::engines_precompiled::nhwc2nchw_params_t<float>, float const*, float*)",
          "duration_us": 5.28,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 4,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 27,
            "registers per thread": 32,
            "shared memory": 4224,
            "blocks per SM": 17.727272,
            "warps per SM": 141.818176,
            "grid": [
              195,
              12,
              1
            ],
            "block": [
              256,
              1,
              1
            ],
            "est. achieved occupancy %": 100,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 8,
              "limitingFactors": "WARPS|REGS",
              "blockLimitRegs": 8,
              "blockLimitSharedMem": 44,
              "blockLimitWarps": 8,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 8192,
              "allocatedSharedMemPerBlock": 5248
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        },
        {
          "name": "void at::native::elementwise_kernel<128, 2, at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<float> >(at::TensorIteratorBase&, at::native::CUDAFunctor_add<float> const&)::{lambda(int)#1}>(int, at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<float> >(at::TensorIteratorBase&, at::native::CUDAFunctor_add<float> const&)::{lambda(int)#1})",
          "duration_us": 7.68,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 7,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 37,
            "registers per thread": 18,
            "shared memory": 0,
            "blocks per SM": 70.909088,
            "warps per SM": 283.636353,
            "grid": [
              9360,
              1,
              1
            ],
            "block": [
              128,
              1,
              1
            ],
            "est. achieved occupancy %": 100,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 16,
              "limitingFactors": "WARPS",
              "blockLimitRegs": 21,
              "blockLimitSharedMem": 228,
              "blockLimitWarps": 16,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 3072,
              "allocatedSharedMemPerBlock": 1024
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        }
      ],
      "checks": {
        "exit_zero": true,
        "input_hash_matches": true,
        "finite_output": true,
        "actual_gpu_kernel_recorded": true,
        "operator_recorded": true
      },
      "evidence_checks_passed": true
    },
    {
      "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_99475e040d65d084.pt",
      "sha256": "0cc0a21ce474602c53c3e5621e64598ed7be5a2c16603fc968441df41de71330",
      "result": {
        "op": "aten.conv3d.default",
        "stage": "vae",
        "shape": [
          1,
          192,
          1,
          240,
          104
        ],
        "exact": true,
        "max_abs": 0.0,
        "rmse": 0.0,
        "finite": true,
        "math_policy": {
          "matmul_precision": "highest",
          "matmul_allow_tf32": false,
          "cudnn_allow_tf32": true,
          "cudnn_benchmark": false,
          "cudnn_deterministic": false
        },
        "scope": "Original ATen op on saved real tensors; isolated diagnostic"
      },
      "trace": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_replay_480x832_v2/014_vae.trace.json",
      "trace_sha256": "97c555c1cea1ae9dfc7712a2ccb7e844fe2fdeb95ea6ab80d554bc14b89ecc97",
      "kernels": [
        {
          "name": "void cudnn::engines_precompiled::nchwToNhwcKernel<float, float, float, false, true, (cudnnKernelDataType_t)0>(cudnn::engines_precompiled::nchw2nhwc_params_t<float>, float const*, float*)",
          "duration_us": 37.343,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 4,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 12,
            "registers per thread": 32,
            "shared memory": 4224,
            "blocks per SM": 109.318184,
            "warps per SM": 874.545471,
            "grid": [
              2405,
              6,
              1
            ],
            "block": [
              256,
              1,
              1
            ],
            "est. achieved occupancy %": 100,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 8,
              "limitingFactors": "WARPS|REGS",
              "blockLimitRegs": 8,
              "blockLimitSharedMem": 44,
              "blockLimitWarps": 8,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 8192,
              "allocatedSharedMemPerBlock": 5248
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        },
        {
          "name": "void cudnn::engines_precompiled::nchwToNhwcKernel<float, float, float, false, true, (cudnnKernelDataType_t)0>(cudnn::engines_precompiled::nchw2nhwc_params_t<float>, float const*, float*)",
          "duration_us": 4.769,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 4,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 16,
            "registers per thread": 32,
            "shared memory": 4224,
            "blocks per SM": 8.727273,
            "warps per SM": 69.818184,
            "grid": [
              1,
              6,
              192
            ],
            "block": [
              256,
              1,
              1
            ],
            "est. achieved occupancy %": 100,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 8,
              "limitingFactors": "WARPS|REGS",
              "blockLimitRegs": 8,
              "blockLimitSharedMem": 44,
              "blockLimitWarps": 8,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 8192,
              "allocatedSharedMemPerBlock": 5248
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        },
        {
          "name": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_k_off_kernel__5x_cudnn",
          "duration_us": 163.84,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 4,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 23,
            "registers per thread": 168,
            "shared memory": 231424,
            "blocks per SM": 1.0,
            "warps per SM": 12.0,
            "grid": [
              132,
              1,
              1
            ],
            "block": [
              384,
              1,
              1
            ],
            "est. achieved occupancy %": 0,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 0,
              "limitingFactors": "SMEM",
              "blockLimitRegs": 1,
              "blockLimitSharedMem": 0,
              "blockLimitWarps": 5,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 64512,
              "allocatedSharedMemPerBlock": 232448
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        },
        {
          "name": "void cudnn::engines_precompiled::nhwcToNchwKernel<float, float, float, true, false, (cudnnKernelDataType_t)0>(cudnn::engines_precompiled::nhwc2nchw_params_t<float>, float const*, float*)",
          "duration_us": 9.249,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 4,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 27,
            "registers per thread": 32,
            "shared memory": 4224,
            "blocks per SM": 35.454544,
            "warps per SM": 283.636353,
            "grid": [
              780,
              6,
              1
            ],
            "block": [
              256,
              1,
              1
            ],
            "est. achieved occupancy %": 100,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 8,
              "limitingFactors": "WARPS|REGS",
              "blockLimitRegs": 8,
              "blockLimitSharedMem": 44,
              "blockLimitWarps": 8,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 8192,
              "allocatedSharedMemPerBlock": 5248
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        },
        {
          "name": "void at::native::elementwise_kernel<128, 2, at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<float> >(at::TensorIteratorBase&, at::native::CUDAFunctor_add<float> const&)::{lambda(int)#1}>(int, at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<float> >(at::TensorIteratorBase&, at::native::CUDAFunctor_add<float> const&)::{lambda(int)#1})",
          "duration_us": 13.215,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 7,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 37,
            "registers per thread": 18,
            "shared memory": 0,
            "blocks per SM": 141.818176,
            "warps per SM": 567.272705,
            "grid": [
              18720,
              1,
              1
            ],
            "block": [
              128,
              1,
              1
            ],
            "est. achieved occupancy %": 100,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 16,
              "limitingFactors": "WARPS",
              "blockLimitRegs": 21,
              "blockLimitSharedMem": 228,
              "blockLimitWarps": 16,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 3072,
              "allocatedSharedMemPerBlock": 1024
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        }
      ],
      "checks": {
        "exit_zero": true,
        "input_hash_matches": true,
        "finite_output": true,
        "actual_gpu_kernel_recorded": true,
        "operator_recorded": true
      },
      "evidence_checks_passed": true
    },
    {
      "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_0bd1bd0a5c44ff3b.pt",
      "sha256": "1e47965836d719f164296b5c8d52cac8c22a10d9ff8c5f4fdfa48498909f8856",
      "result": {
        "op": "aten.conv3d.default",
        "stage": "vae",
        "shape": [
          1,
          96,
          1,
          480,
          208
        ],
        "exact": true,
        "max_abs": 0.0,
        "rmse": 0.0,
        "finite": true,
        "math_policy": {
          "matmul_precision": "highest",
          "matmul_allow_tf32": false,
          "cudnn_allow_tf32": true,
          "cudnn_benchmark": false,
          "cudnn_deterministic": false
        },
        "scope": "Original ATen op on saved real tensors; isolated diagnostic"
      },
      "trace": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_replay_480x832_v2/015_vae.trace.json",
      "trace_sha256": "6d47b822d1f005d7a17baec0d6c08f53687391c779ad9e06dca571464ead28e8",
      "kernels": [
        {
          "name": "void cudnn::engines_precompiled::nchwToNhwcKernel<float, float, float, false, true, (cudnnKernelDataType_t)0>(cudnn::engines_precompiled::nchw2nhwc_params_t<float>, float const*, float*)",
          "duration_us": 73.216,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 4,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 12,
            "registers per thread": 32,
            "shared memory": 4224,
            "blocks per SM": 215.681824,
            "warps per SM": 1725.45459,
            "grid": [
              9490,
              3,
              1
            ],
            "block": [
              256,
              1,
              1
            ],
            "est. achieved occupancy %": 100,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 8,
              "limitingFactors": "WARPS|REGS",
              "blockLimitRegs": 8,
              "blockLimitSharedMem": 44,
              "blockLimitWarps": 8,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 8192,
              "allocatedSharedMemPerBlock": 5248
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        },
        {
          "name": "void cudnn::engines_precompiled::nchwToNhwcKernel<float, float, float, false, true, (cudnnKernelDataType_t)0>(cudnn::engines_precompiled::nchw2nhwc_params_t<float>, float const*, float*)",
          "duration_us": 2.688,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 4,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 16,
            "registers per thread": 32,
            "shared memory": 4224,
            "blocks per SM": 2.181818,
            "warps per SM": 17.454546,
            "grid": [
              1,
              3,
              96
            ],
            "block": [
              256,
              1,
              1
            ],
            "est. achieved occupancy %": 27,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 8,
              "limitingFactors": "WARPS|REGS",
              "blockLimitRegs": 8,
              "blockLimitSharedMem": 44,
              "blockLimitWarps": 8,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 8192,
              "allocatedSharedMemPerBlock": 5248
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        },
        {
          "name": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize256x32x32_warpgroupsize1x1x1_g1_execute_segment_k_off_kernel__5x_cudnn",
          "duration_us": 344.96,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 4,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 23,
            "registers per thread": 168,
            "shared memory": 231424,
            "blocks per SM": 1.0,
            "warps per SM": 12.0,
            "grid": [
              66,
              2,
              1
            ],
            "block": [
              384,
              1,
              1
            ],
            "est. achieved occupancy %": 0,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 0,
              "limitingFactors": "SMEM",
              "blockLimitRegs": 1,
              "blockLimitSharedMem": 0,
              "blockLimitWarps": 5,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 64512,
              "allocatedSharedMemPerBlock": 232448
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        },
        {
          "name": "void cudnn::engines_precompiled::nhwcToNchwKernel<float, float, float, true, false, (cudnnKernelDataType_t)0>(cudnn::engines_precompiled::nhwc2nchw_params_t<float>, float const*, float*)",
          "duration_us": 19.393,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 4,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 27,
            "registers per thread": 32,
            "shared memory": 4224,
            "blocks per SM": 70.909088,
            "warps per SM": 567.272705,
            "grid": [
              3120,
              3,
              1
            ],
            "block": [
              256,
              1,
              1
            ],
            "est. achieved occupancy %": 100,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 8,
              "limitingFactors": "WARPS|REGS",
              "blockLimitRegs": 8,
              "blockLimitSharedMem": 44,
              "blockLimitWarps": 8,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 8192,
              "allocatedSharedMemPerBlock": 5248
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        },
        {
          "name": "void at::native::elementwise_kernel<128, 2, at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<float> >(at::TensorIteratorBase&, at::native::CUDAFunctor_add<float> const&)::{lambda(int)#1}>(int, at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<float> >(at::TensorIteratorBase&, at::native::CUDAFunctor_add<float> const&)::{lambda(int)#1})",
          "duration_us": 27.84,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 7,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 37,
            "registers per thread": 18,
            "shared memory": 0,
            "blocks per SM": 283.636353,
            "warps per SM": 1134.54541,
            "grid": [
              37440,
              1,
              1
            ],
            "block": [
              128,
              1,
              1
            ],
            "est. achieved occupancy %": 100,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 16,
              "limitingFactors": "WARPS",
              "blockLimitRegs": 21,
              "blockLimitSharedMem": 228,
              "blockLimitWarps": 16,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 3072,
              "allocatedSharedMemPerBlock": 1024
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        }
      ],
      "checks": {
        "exit_zero": true,
        "input_hash_matches": true,
        "finite_output": true,
        "actual_gpu_kernel_recorded": true,
        "operator_recorded": true
      },
      "evidence_checks_passed": true
    },
    {
      "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/output_rank0_f18eb91117dda1c8.pt",
      "sha256": "897dd2bd3ad5fe7258ee5c3ae9cfd93da4e5e94644902267f0b3d8bd156f63b0",
      "result": {
        "op": "aten.to.dtype",
        "stage": "output",
        "shape": [
          1,
          3,
          9,
          480,
          832
        ],
        "exact": true,
        "max_abs": 0.0,
        "rmse": 0.0,
        "finite": true,
        "math_policy": {
          "matmul_precision": "highest",
          "matmul_allow_tf32": false,
          "cudnn_allow_tf32": true,
          "cudnn_benchmark": false,
          "cudnn_deterministic": false
        },
        "scope": "Original ATen op on saved real tensors; isolated diagnostic"
      },
      "trace": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_replay_480x832_v2/016_output.trace.json",
      "trace_sha256": "a3559b906e1f6675c7c04cc5a0a249501ee447441da27f63b0dc30d56bd249d4",
      "kernels": [
        {
          "name": "void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda()#3}::operator()() const::{lambda()#1}::operator()() const::{lambda(unsigned char)#1}, std::array<char*, 2ul>, 4, TrivialOffsetCalculator<1, unsigned int>, TrivialOffsetCalculator<1, unsigned int>, at::native::memory::LoadWithCast<1>, at::native::memory::StoreWithCast<1> >(int, at::native::direct_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda()#3}::operator()() const::{lambda()#1}::operator()() const::{lambda(unsigned char)#1}, std::array<char*, 2ul>, TrivialOffsetCalculator<1, unsigned int>, TrivialOffsetCalculator<1, unsigned int>, at::native::memory::LoadWithCast<1>, at::native::memory::StoreWithCast<1>)",
          "duration_us": 31.265,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 4,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 11,
            "registers per thread": 32,
            "shared memory": 0,
            "blocks per SM": 159.545456,
            "warps per SM": 638.181824,
            "grid": [
              21060,
              1,
              1
            ],
            "block": [
              128,
              1,
              1
            ],
            "est. achieved occupancy %": 100,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 16,
              "limitingFactors": "WARPS|REGS",
              "blockLimitRegs": 16,
              "blockLimitSharedMem": 228,
              "blockLimitWarps": 16,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 4096,
              "allocatedSharedMemPerBlock": 1024
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        }
      ],
      "checks": {
        "exit_zero": true,
        "input_hash_matches": true,
        "finite_output": true,
        "actual_gpu_kernel_recorded": true,
        "operator_recorded": true
      },
      "evidence_checks_passed": true
    },
    {
      "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/output_rank0_e753c57113db57aa.pt",
      "sha256": "a9aab61b63e7cab483e81582c9af41d89d525814ab14de753e8a7b16de6c4f1b",
      "result": {
        "op": "aten.to.dtype",
        "stage": "output",
        "shape": [
          1,
          3,
          12,
          480,
          832
        ],
        "exact": true,
        "max_abs": 0.0,
        "rmse": 0.0,
        "finite": true,
        "math_policy": {
          "matmul_precision": "highest",
          "matmul_allow_tf32": false,
          "cudnn_allow_tf32": true,
          "cudnn_benchmark": false,
          "cudnn_deterministic": false
        },
        "scope": "Original ATen op on saved real tensors; isolated diagnostic"
      },
      "trace": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_replay_480x832_v2/017_output.trace.json",
      "trace_sha256": "46b9179267d5c728836b8e2d658454a756add25b11bc36eea8cffda2b96bd7a7",
      "kernels": [
        {
          "name": "void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda()#3}::operator()() const::{lambda()#1}::operator()() const::{lambda(unsigned char)#1}, std::array<char*, 2ul>, 4, TrivialOffsetCalculator<1, unsigned int>, TrivialOffsetCalculator<1, unsigned int>, at::native::memory::LoadWithCast<1>, at::native::memory::StoreWithCast<1> >(int, at::native::direct_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda()#3}::operator()() const::{lambda()#1}::operator()() const::{lambda(unsigned char)#1}, std::array<char*, 2ul>, TrivialOffsetCalculator<1, unsigned int>, TrivialOffsetCalculator<1, unsigned int>, at::native::memory::LoadWithCast<1>, at::native::memory::StoreWithCast<1>)",
          "duration_us": 40.288,
          "pid": 0,
          "tid": 7,
          "args": {
            "External id": 4,
            "queued": 0,
            "device": 0,
            "context": 1,
            "stream": 7,
            "correlation": 11,
            "registers per thread": 32,
            "shared memory": 0,
            "blocks per SM": 212.72728,
            "warps per SM": 850.909119,
            "grid": [
              28080,
              1,
              1
            ],
            "block": [
              128,
              1,
              1
            ],
            "est. achieved occupancy %": 100,
            "occupancy": {
              "activeBlocksPerMultiprocessor": 16,
              "limitingFactors": "WARPS|REGS",
              "blockLimitRegs": 16,
              "blockLimitSharedMem": 228,
              "blockLimitWarps": 16,
              "blockLimitBlocks": 32,
              "blockLimitBarriers": 2147483647,
              "allocatedRegistersPerBlock": 4096,
              "allocatedSharedMemPerBlock": 1024
            },
            "graph id": 0,
            "graph node id": 0,
            "channel": 0,
            "channel_type": 1
          }
        }
      ],
      "checks": {
        "exit_zero": true,
        "input_hash_matches": true,
        "finite_output": true,
        "actual_gpu_kernel_recorded": true,
        "operator_recorded": true
      },
      "evidence_checks_passed": true
    }
  ],
  "nonexact_output_count": 0,
  "hardware_counter_status": "missing_for_these_operator_inputs"
}
