{
  "scope": "18 saved rank0 operator replays versus latest worker-flush steady launch metadata. Signature equality uses full name, grid, block, registers and total shared bytes. Neither match nor a unique captured candidate assigns an exact model shape; multiple saved inputs can share a signature, and missing matches can reflect compilation/algorithm differences. Summed model time is across GPUs, not wall time or replay timing.",
  "source_sha256": {
    "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/analysis/operator_replay_v2_audit.json": "f2daaae4ccf2a02d7e865a596e28169322898abbe1b21266a68dd72c5b8ea0a1",
    "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/analysis/nsys_worker_flush_v2_physical/launch_resources_by_physical_gpu_steady.csv": "8affd589e3c7c1bb4faa2c1be91709aadd9478f2a952e7a1f1bdff0c7b2549a9"
  },
  "replay_inputs": 18,
  "replay_signatures": 35,
  "matched_signatures": 10,
  "ambiguous_matched_signatures": 3,
  "matched_signatures_with_distinct_output_shapes": 1,
  "rows": [
    {
      "name": "nvjet_sm90_tst_256x144_64x4_2x1_v_bz_coopA_bias_TNT",
      "grid": [
        2,
        66,
        1
      ],
      "block": [
        384,
        1,
        1
      ],
      "registers_per_thread": 168,
      "total_shared_memory_bytes": 221380,
      "saved_input_count": 3,
      "distinct_output_shapes": [
        [
          1,
          4680,
          5120
        ],
        [
          4680,
          5120
        ]
      ],
      "ambiguous_among_saved_inputs": true,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/dit_rank0_cd8cbc742f5c3f37.pt",
          "input_sha256": "824a77bcaead8af0dacf908827027966a81c2d964d919eb4200bac0771b4a369",
          "op": "aten.linear.default",
          "stage": "dit",
          "output_shape": [
            4680,
            5120
          ]
        },
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/dit_rank0_9296b219534b05e0.pt",
          "input_sha256": "787b43c2c5ab957d3bd3c45047d646e6e407e506ea71b3249408d5d20bf00b0c",
          "op": "aten.linear.default",
          "stage": "dit",
          "output_shape": [
            1,
            4680,
            5120
          ]
        },
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/dit_rank0_f56603026672560c.pt",
          "input_sha256": "4a5b492ee606c1a2c9fe4844a0c3125609fa599f4c027f56b4329c89d2879b7e",
          "op": "aten.linear.default",
          "stage": "dit",
          "output_shape": [
            1,
            4680,
            5120
          ]
        }
      ],
      "model_physical_groups": 4,
      "model_steady_calls": 320,
      "model_clipped_summed_gpu_ms": 58.543511,
      "model_first_event_ids": [
        {
          "physical_gpu": "3",
          "first_event_id": "870765"
        },
        {
          "physical_gpu": "2",
          "first_event_id": "870764"
        },
        {
          "physical_gpu": "0",
          "first_event_id": "870768"
        },
        {
          "physical_gpu": "1",
          "first_event_id": "870766"
        }
      ],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "nvjet_sm90_tst_64x8_64x16_4x1_v_bz_bias_TNT",
      "grid": [
        4,
        20,
        1
      ],
      "block": [
        384,
        1,
        1
      ],
      "registers_per_thread": 168,
      "total_shared_memory_bytes": 164308,
      "saved_input_count": 2,
      "distinct_output_shapes": [
        [
          3,
          5120
        ]
      ],
      "ambiguous_among_saved_inputs": true,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/dit_rank0_10a11d3772070763.pt",
          "input_sha256": "ea1184eaad12555ac80bf5ee33de7aa2251926b289572cadadb4043884848587",
          "op": "aten.linear.default",
          "stage": "dit",
          "output_shape": [
            3,
            5120
          ]
        },
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/dit_rank0_ca590b3a668a9ecf.pt",
          "input_sha256": "a6cb8d5375b991c9c5d1a1e5674fa86eb61d10648b2ccf71bb7ec40c56f2ac1e",
          "op": "aten.linear.default",
          "stage": "dit",
          "output_shape": [
            3,
            5120
          ]
        }
      ],
      "model_physical_groups": 4,
      "model_steady_calls": 160,
      "model_clipped_summed_gpu_ms": 1.928162,
      "model_first_event_ids": [
        {
          "physical_gpu": "3",
          "first_event_id": "870831"
        },
        {
          "physical_gpu": "1",
          "first_event_id": "870828"
        },
        {
          "physical_gpu": "2",
          "first_event_id": "870802"
        },
        {
          "physical_gpu": "0",
          "first_event_id": "870851"
        }
      ],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "nvjet_sm90_tst_64x8_64x16_2x1_v_bz_bias_TNT",
      "grid": [
        2,
        66,
        1
      ],
      "block": [
        384,
        1,
        1
      ],
      "registers_per_thread": 168,
      "total_shared_memory_bytes": 164308,
      "saved_input_count": 1,
      "distinct_output_shapes": [
        [
          1,
          3,
          30720
        ]
      ],
      "ambiguous_among_saved_inputs": false,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/dit_rank0_22805ed4eb786e66.pt",
          "input_sha256": "d6a2e382ecf0fdad711df2cd1bcb239a004321ae666d26aa04920718cf49bf81",
          "op": "aten.linear.default",
          "stage": "dit",
          "output_shape": [
            1,
            3,
            30720
          ]
        }
      ],
      "model_physical_groups": 4,
      "model_steady_calls": 80,
      "model_clipped_summed_gpu_ms": 6.190926,
      "model_first_event_ids": [
        {
          "physical_gpu": "2",
          "first_event_id": "870830"
        },
        {
          "physical_gpu": "0",
          "first_event_id": "870855"
        },
        {
          "physical_gpu": "3",
          "first_event_id": "870848"
        },
        {
          "physical_gpu": "1",
          "first_event_id": "870846"
        }
      ],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "nvjet_sm90_tst_256x152_64x4_1x2_h_bz_coopA_bias_TNT",
      "grid": [
        2,
        66,
        1
      ],
      "block": [
        384,
        1,
        1
      ],
      "registers_per_thread": 168,
      "total_shared_memory_bytes": 225476,
      "saved_input_count": 1,
      "distinct_output_shapes": [
        [
          1,
          1170,
          15360
        ]
      ],
      "ambiguous_among_saved_inputs": false,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/dit_rank0_fcebe966800b94b1.pt",
          "input_sha256": "4f1e9988bf86b3f91f274f9b8c3bf33bf42b39945181ce6f09fbce8e0132cb6f",
          "op": "aten.linear.default",
          "stage": "dit",
          "output_shape": [
            1,
            1170,
            15360
          ]
        }
      ],
      "model_physical_groups": 4,
      "model_steady_calls": 3200,
      "model_clipped_summed_gpu_ms": 853.086899,
      "model_first_event_ids": [
        {
          "physical_gpu": "3",
          "first_event_id": "870864"
        },
        {
          "physical_gpu": "2",
          "first_event_id": "870861"
        },
        {
          "physical_gpu": "1",
          "first_event_id": "870867"
        },
        {
          "physical_gpu": "0",
          "first_event_id": "870866"
        }
      ],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "nvjet_sm90_tst_128x200_64x5_2x1_v_bz_coopA_bias_TNT",
      "grid": [
        2,
        66,
        1
      ],
      "block": [
        384,
        1,
        1
      ],
      "registers_per_thread": 168,
      "total_shared_memory_bytes": 226492,
      "saved_input_count": 1,
      "distinct_output_shapes": [
        [
          1,
          1170,
          5120
        ]
      ],
      "ambiguous_among_saved_inputs": false,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/dit_rank0_10a9e2fea034c6cf.pt",
          "input_sha256": "a64f7da6eb1fcc9779d40258d974eb529c2397aaba4df6e32363d7c5d4e73065",
          "op": "aten.linear.default",
          "stage": "dit",
          "output_shape": [
            1,
            1170,
            5120
          ]
        }
      ],
      "model_physical_groups": 4,
      "model_steady_calls": 28800,
      "model_clipped_summed_gpu_ms": 3432.218463,
      "model_first_event_ids": [
        {
          "physical_gpu": "3",
          "first_event_id": "870934"
        },
        {
          "physical_gpu": "0",
          "first_event_id": "870931"
        },
        {
          "physical_gpu": "1",
          "first_event_id": "870929"
        },
        {
          "physical_gpu": "2",
          "first_event_id": "870935"
        }
      ],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "void cutlass::Kernel2<cutlass_80_tensorop_s1688gemm_128x64_16x6_nn_align4>(cutlass_80_tensorop_s1688gemm_128x64_16x6_nn_align4::Params)",
      "grid": [
        8,
        19,
        1
      ],
      "block": [
        128,
        1,
        1
      ],
      "registers_per_thread": 136,
      "total_shared_memory_bytes": 73728,
      "saved_input_count": 1,
      "distinct_output_shapes": [
        [
          1,
          16,
          3,
          60,
          104
        ]
      ],
      "ambiguous_among_saved_inputs": false,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_6638f430ac58a438.pt",
          "input_sha256": "32bcd9b4d8be26c876626649d4ff2e623424efe309222bcd7e80afc47310ded4",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            16,
            3,
            60,
            104
          ]
        }
      ],
      "model_physical_groups": 0,
      "model_steady_calls": 0,
      "model_clipped_summed_gpu_ms": 0,
      "model_first_event_ids": [],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "void at::native::elementwise_kernel<128, 2, at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<float> >(at::TensorIteratorBase&, at::native::CUDAFunctor_add<float> const&)::{lambda(int)#1}>(int, at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<float> >(at::TensorIteratorBase&, at::native::CUDAFunctor_add<float> const&)::{lambda(int)#1})",
      "grid": [
        1170,
        1,
        1
      ],
      "block": [
        128,
        1,
        1
      ],
      "registers_per_thread": 18,
      "total_shared_memory_bytes": 0,
      "saved_input_count": 1,
      "distinct_output_shapes": [
        [
          1,
          16,
          3,
          60,
          104
        ]
      ],
      "ambiguous_among_saved_inputs": false,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_6638f430ac58a438.pt",
          "input_sha256": "32bcd9b4d8be26c876626649d4ff2e623424efe309222bcd7e80afc47310ded4",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            16,
            3,
            60,
            104
          ]
        }
      ],
      "model_physical_groups": 0,
      "model_steady_calls": 0,
      "model_clipped_summed_gpu_ms": 0,
      "model_first_event_ids": [],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "void cudnn::engines_precompiled::nchwToNhwcKernel<float, float, float, false, true, (cudnnKernelDataType_t)2>(cudnn::engines_precompiled::nchw2nhwc_params_t<float>, float const*, float*)",
      "grid": [
        163,
        1,
        1
      ],
      "block": [
        256,
        1,
        1
      ],
      "registers_per_thread": 32,
      "total_shared_memory_bytes": 4224,
      "saved_input_count": 1,
      "distinct_output_shapes": [
        [
          1,
          384,
          1,
          60,
          26
        ]
      ],
      "ambiguous_among_saved_inputs": false,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_29ef146661136bf4.pt",
          "input_sha256": "2b9e80cee384305d79045df290612b04bcfe041bd67dc69a2bc096e45a6c5872",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            384,
            1,
            60,
            26
          ]
        }
      ],
      "model_physical_groups": 0,
      "model_steady_calls": 0,
      "model_clipped_summed_gpu_ms": 0,
      "model_first_event_ids": [],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "void cudnn::engines_precompiled::nchwToNhwcKernel<float, float, float, false, true, (cudnnKernelDataType_t)2>(cudnn::engines_precompiled::nchw2nhwc_params_t<float>, float const*, float*)",
      "grid": [
        1,
        1,
        384
      ],
      "block": [
        256,
        1,
        1
      ],
      "registers_per_thread": 32,
      "total_shared_memory_bytes": 4224,
      "saved_input_count": 1,
      "distinct_output_shapes": [
        [
          1,
          384,
          1,
          60,
          26
        ]
      ],
      "ambiguous_among_saved_inputs": false,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_29ef146661136bf4.pt",
          "input_sha256": "2b9e80cee384305d79045df290612b04bcfe041bd67dc69a2bc096e45a6c5872",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            384,
            1,
            60,
            26
          ]
        }
      ],
      "model_physical_groups": 0,
      "model_steady_calls": 0,
      "model_clipped_summed_gpu_ms": 0,
      "model_first_event_ids": [],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "sm80_xmma_fprop_implicit_gemm_tf32f32_tf32f32_f32_nhwckrsc_nchw_tilesize128x64x32_stage5_warpsize2x2x1_g1_tensor16x8x8_execute_kernel__5x_cudnn",
      "grid": [
        6,
        13,
        1
      ],
      "block": [
        128,
        1,
        1
      ],
      "registers_per_thread": 168,
      "total_shared_memory_bytes": 122880,
      "saved_input_count": 1,
      "distinct_output_shapes": [
        [
          1,
          384,
          1,
          60,
          26
        ]
      ],
      "ambiguous_among_saved_inputs": false,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_29ef146661136bf4.pt",
          "input_sha256": "2b9e80cee384305d79045df290612b04bcfe041bd67dc69a2bc096e45a6c5872",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            384,
            1,
            60,
            26
          ]
        }
      ],
      "model_physical_groups": 4,
      "model_steady_calls": 48,
      "model_clipped_summed_gpu_ms": 0.767361,
      "model_first_event_ids": [
        {
          "physical_gpu": "2",
          "first_event_id": "905371"
        },
        {
          "physical_gpu": "1",
          "first_event_id": "905408"
        },
        {
          "physical_gpu": "0",
          "first_event_id": "905370"
        },
        {
          "physical_gpu": "3",
          "first_event_id": "905356"
        }
      ],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "void at::native::elementwise_kernel<128, 2, at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<float> >(at::TensorIteratorBase&, at::native::CUDAFunctor_add<float> const&)::{lambda(int)#1}>(int, at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<float> >(at::TensorIteratorBase&, at::native::CUDAFunctor_add<float> const&)::{lambda(int)#1})",
      "grid": [
        2340,
        1,
        1
      ],
      "block": [
        128,
        1,
        1
      ],
      "registers_per_thread": 18,
      "total_shared_memory_bytes": 0,
      "saved_input_count": 2,
      "distinct_output_shapes": [
        [
          1,
          384,
          1,
          60,
          26
        ]
      ],
      "ambiguous_among_saved_inputs": true,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_29ef146661136bf4.pt",
          "input_sha256": "2b9e80cee384305d79045df290612b04bcfe041bd67dc69a2bc096e45a6c5872",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            384,
            1,
            60,
            26
          ]
        },
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_551fa1ff3095f40c.pt",
          "input_sha256": "6309c12f24013edabd88edced7c2ce22672322d203206b12ba218c88696effb5",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            384,
            1,
            60,
            26
          ]
        }
      ],
      "model_physical_groups": 0,
      "model_steady_calls": 0,
      "model_clipped_summed_gpu_ms": 0,
      "model_first_event_ids": [],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "void cudnn::engines_precompiled::nchwToNhwcKernel<float, float, float, false, true, (cudnnKernelDataType_t)0>(cudnn::engines_precompiled::nchw2nhwc_params_t<float>, float const*, float*)",
      "grid": [
        163,
        12,
        1
      ],
      "block": [
        256,
        1,
        1
      ],
      "registers_per_thread": 32,
      "total_shared_memory_bytes": 4224,
      "saved_input_count": 1,
      "distinct_output_shapes": [
        [
          1,
          384,
          1,
          60,
          26
        ]
      ],
      "ambiguous_among_saved_inputs": false,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_551fa1ff3095f40c.pt",
          "input_sha256": "6309c12f24013edabd88edced7c2ce22672322d203206b12ba218c88696effb5",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            384,
            1,
            60,
            26
          ]
        }
      ],
      "model_physical_groups": 0,
      "model_steady_calls": 0,
      "model_clipped_summed_gpu_ms": 0,
      "model_first_event_ids": [],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "void cudnn::engines_precompiled::nchwToNhwcKernel<float, float, float, false, true, (cudnnKernelDataType_t)0>(cudnn::engines_precompiled::nchw2nhwc_params_t<float>, float const*, float*)",
      "grid": [
        1,
        12,
        384
      ],
      "block": [
        256,
        1,
        1
      ],
      "registers_per_thread": 32,
      "total_shared_memory_bytes": 4224,
      "saved_input_count": 2,
      "distinct_output_shapes": [
        [
          1,
          384,
          1,
          60,
          26
        ],
        [
          1,
          384,
          1,
          120,
          52
        ]
      ],
      "ambiguous_among_saved_inputs": true,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_551fa1ff3095f40c.pt",
          "input_sha256": "6309c12f24013edabd88edced7c2ce22672322d203206b12ba218c88696effb5",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            384,
            1,
            60,
            26
          ]
        },
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_8601bc3a88374d99.pt",
          "input_sha256": "c7f10fe8b8ca84229e785645c70c003456c3950c79cdd176694e5bc86c9247f4",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            384,
            1,
            120,
            52
          ]
        }
      ],
      "model_physical_groups": 0,
      "model_steady_calls": 0,
      "model_clipped_summed_gpu_ms": 0,
      "model_first_event_ids": [],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize64x64x32_warpgroupsize1x1x1_g1_execute_segment_k_off_kernel__5x_cudnn",
      "grid": [
        60,
        2,
        1
      ],
      "block": [
        384,
        1,
        1
      ],
      "registers_per_thread": 168,
      "total_shared_memory_bytes": 231424,
      "saved_input_count": 1,
      "distinct_output_shapes": [
        [
          1,
          384,
          1,
          60,
          26
        ]
      ],
      "ambiguous_among_saved_inputs": false,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_551fa1ff3095f40c.pt",
          "input_sha256": "6309c12f24013edabd88edced7c2ce22672322d203206b12ba218c88696effb5",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            384,
            1,
            60,
            26
          ]
        }
      ],
      "model_physical_groups": 4,
      "model_steady_calls": 480,
      "model_clipped_summed_gpu_ms": 45.933336,
      "model_first_event_ids": [
        {
          "physical_gpu": "3",
          "first_event_id": "905439"
        },
        {
          "physical_gpu": "2",
          "first_event_id": "905478"
        },
        {
          "physical_gpu": "0",
          "first_event_id": "905479"
        },
        {
          "physical_gpu": "1",
          "first_event_id": "905527"
        }
      ],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "void cudnn::engines_precompiled::nhwcToNchwKernel<float, float, float, true, false, (cudnnKernelDataType_t)0>(cudnn::engines_precompiled::nhwc2nchw_params_t<float>, float const*, float*)",
      "grid": [
        49,
        12,
        1
      ],
      "block": [
        256,
        1,
        1
      ],
      "registers_per_thread": 32,
      "total_shared_memory_bytes": 4224,
      "saved_input_count": 1,
      "distinct_output_shapes": [
        [
          1,
          384,
          1,
          60,
          26
        ]
      ],
      "ambiguous_among_saved_inputs": false,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_551fa1ff3095f40c.pt",
          "input_sha256": "6309c12f24013edabd88edced7c2ce22672322d203206b12ba218c88696effb5",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            384,
            1,
            60,
            26
          ]
        }
      ],
      "model_physical_groups": 0,
      "model_steady_calls": 0,
      "model_clipped_summed_gpu_ms": 0,
      "model_first_event_ids": [],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "void cutlass::Kernel2<cutlass_80_tensorop_s1688gemm_64x128_32x3_nn_align4>(cutlass_80_tensorop_s1688gemm_64x128_32x3_nn_align4::Params)",
      "grid": [
        24,
        13,
        1
      ],
      "block": [
        128,
        1,
        1
      ],
      "registers_per_thread": 168,
      "total_shared_memory_bytes": 73728,
      "saved_input_count": 1,
      "distinct_output_shapes": [
        [
          1,
          384,
          1,
          120,
          52
        ]
      ],
      "ambiguous_among_saved_inputs": false,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_c16d11a7624e1a8c.pt",
          "input_sha256": "1e43418295ca6422f27afcf0b9940aa818fbb12fcf1013c12dfcf7a60f86e7d7",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            384,
            1,
            120,
            52
          ]
        }
      ],
      "model_physical_groups": 0,
      "model_steady_calls": 0,
      "model_clipped_summed_gpu_ms": 0,
      "model_first_event_ids": [],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "void at::native::elementwise_kernel<128, 2, at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<float> >(at::TensorIteratorBase&, at::native::CUDAFunctor_add<float> const&)::{lambda(int)#1}>(int, at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<float> >(at::TensorIteratorBase&, at::native::CUDAFunctor_add<float> const&)::{lambda(int)#1})",
      "grid": [
        9360,
        1,
        1
      ],
      "block": [
        128,
        1,
        1
      ],
      "registers_per_thread": 18,
      "total_shared_memory_bytes": 0,
      "saved_input_count": 3,
      "distinct_output_shapes": [
        [
          1,
          384,
          1,
          120,
          52
        ]
      ],
      "ambiguous_among_saved_inputs": true,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_c16d11a7624e1a8c.pt",
          "input_sha256": "1e43418295ca6422f27afcf0b9940aa818fbb12fcf1013c12dfcf7a60f86e7d7",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            384,
            1,
            120,
            52
          ]
        },
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_081e405bd1e07dce.pt",
          "input_sha256": "3193a32ddc024e9d34eb5446f9909408dde38d963ebc5cb9654998ecf661a820",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            384,
            1,
            120,
            52
          ]
        },
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_8601bc3a88374d99.pt",
          "input_sha256": "c7f10fe8b8ca84229e785645c70c003456c3950c79cdd176694e5bc86c9247f4",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            384,
            1,
            120,
            52
          ]
        }
      ],
      "model_physical_groups": 0,
      "model_steady_calls": 0,
      "model_clipped_summed_gpu_ms": 0,
      "model_first_event_ids": [],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "void cudnn::engines_precompiled::nchwToNhwcKernel<float, float, float, false, true, (cudnnKernelDataType_t)0>(cudnn::engines_precompiled::nchw2nhwc_params_t<float>, float const*, float*)",
      "grid": [
        618,
        6,
        1
      ],
      "block": [
        256,
        1,
        1
      ],
      "registers_per_thread": 32,
      "total_shared_memory_bytes": 4224,
      "saved_input_count": 1,
      "distinct_output_shapes": [
        [
          1,
          384,
          1,
          120,
          52
        ]
      ],
      "ambiguous_among_saved_inputs": false,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_081e405bd1e07dce.pt",
          "input_sha256": "3193a32ddc024e9d34eb5446f9909408dde38d963ebc5cb9654998ecf661a820",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            384,
            1,
            120,
            52
          ]
        }
      ],
      "model_physical_groups": 0,
      "model_steady_calls": 0,
      "model_clipped_summed_gpu_ms": 0,
      "model_first_event_ids": [],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "void cudnn::engines_precompiled::nchwToNhwcKernel<float, float, float, false, true, (cudnnKernelDataType_t)0>(cudnn::engines_precompiled::nchw2nhwc_params_t<float>, float const*, float*)",
      "grid": [
        1,
        6,
        384
      ],
      "block": [
        256,
        1,
        1
      ],
      "registers_per_thread": 32,
      "total_shared_memory_bytes": 4224,
      "saved_input_count": 1,
      "distinct_output_shapes": [
        [
          1,
          384,
          1,
          120,
          52
        ]
      ],
      "ambiguous_among_saved_inputs": false,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_081e405bd1e07dce.pt",
          "input_sha256": "3193a32ddc024e9d34eb5446f9909408dde38d963ebc5cb9654998ecf661a820",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            384,
            1,
            120,
            52
          ]
        }
      ],
      "model_physical_groups": 0,
      "model_steady_calls": 0,
      "model_clipped_summed_gpu_ms": 0,
      "model_first_event_ids": [],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "void cask_plugin__5x_cudnn::xmma__5x_cudnn::init_device_workspace_kernel<xmma__5x_cudnn::implicit_gemm::fprop::Warp_specialized_params_non_template<xmma__5x_cudnn::Grid_constant_params> >(xmma__5x_cudnn::implicit_gemm::fprop::Warp_specialized_params_non_template<xmma__5x_cudnn::Grid_constant_params>, bool)",
      "grid": [
        1,
        1,
        1
      ],
      "block": [
        1,
        1,
        1
      ],
      "registers_per_thread": 16,
      "total_shared_memory_bytes": 0,
      "saved_input_count": 2,
      "distinct_output_shapes": [
        [
          1,
          384,
          1,
          120,
          52
        ]
      ],
      "ambiguous_among_saved_inputs": true,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_081e405bd1e07dce.pt",
          "input_sha256": "3193a32ddc024e9d34eb5446f9909408dde38d963ebc5cb9654998ecf661a820",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            384,
            1,
            120,
            52
          ]
        },
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_8601bc3a88374d99.pt",
          "input_sha256": "c7f10fe8b8ca84229e785645c70c003456c3950c79cdd176694e5bc86c9247f4",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            384,
            1,
            120,
            52
          ]
        }
      ],
      "model_physical_groups": 0,
      "model_steady_calls": 0,
      "model_clipped_summed_gpu_ms": 0,
      "model_first_event_ids": [],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_k_on_kernel__5x_cudnn",
      "grid": [
        132,
        1,
        1
      ],
      "block": [
        384,
        1,
        1
      ],
      "registers_per_thread": 168,
      "total_shared_memory_bytes": 231424,
      "saved_input_count": 2,
      "distinct_output_shapes": [
        [
          1,
          384,
          1,
          120,
          52
        ]
      ],
      "ambiguous_among_saved_inputs": true,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_081e405bd1e07dce.pt",
          "input_sha256": "3193a32ddc024e9d34eb5446f9909408dde38d963ebc5cb9654998ecf661a820",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            384,
            1,
            120,
            52
          ]
        },
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_8601bc3a88374d99.pt",
          "input_sha256": "c7f10fe8b8ca84229e785645c70c003456c3950c79cdd176694e5bc86c9247f4",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            384,
            1,
            120,
            52
          ]
        }
      ],
      "model_physical_groups": 4,
      "model_steady_calls": 48,
      "model_clipped_summed_gpu_ms": 3.864935,
      "model_first_event_ids": [
        {
          "physical_gpu": "0",
          "first_event_id": "907107"
        },
        {
          "physical_gpu": "1",
          "first_event_id": "907113"
        },
        {
          "physical_gpu": "3",
          "first_event_id": "907101"
        },
        {
          "physical_gpu": "2",
          "first_event_id": "907103"
        }
      ],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "void cudnn::engines_precompiled::nhwcToNchwKernel<float, float, float, true, false, (cudnnKernelDataType_t)0>(cudnn::engines_precompiled::nhwc2nchw_params_t<float>, float const*, float*)",
      "grid": [
        195,
        12,
        1
      ],
      "block": [
        256,
        1,
        1
      ],
      "registers_per_thread": 32,
      "total_shared_memory_bytes": 4224,
      "saved_input_count": 2,
      "distinct_output_shapes": [
        [
          1,
          384,
          1,
          120,
          52
        ]
      ],
      "ambiguous_among_saved_inputs": true,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_081e405bd1e07dce.pt",
          "input_sha256": "3193a32ddc024e9d34eb5446f9909408dde38d963ebc5cb9654998ecf661a820",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            384,
            1,
            120,
            52
          ]
        },
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_8601bc3a88374d99.pt",
          "input_sha256": "c7f10fe8b8ca84229e785645c70c003456c3950c79cdd176694e5bc86c9247f4",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            384,
            1,
            120,
            52
          ]
        }
      ],
      "model_physical_groups": 0,
      "model_steady_calls": 0,
      "model_clipped_summed_gpu_ms": 0,
      "model_first_event_ids": [],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "void cudnn::engines_precompiled::nchwToNhwcKernel<float, float, float, false, true, (cudnnKernelDataType_t)0>(cudnn::engines_precompiled::nchw2nhwc_params_t<float>, float const*, float*)",
      "grid": [
        618,
        12,
        1
      ],
      "block": [
        256,
        1,
        1
      ],
      "registers_per_thread": 32,
      "total_shared_memory_bytes": 4224,
      "saved_input_count": 1,
      "distinct_output_shapes": [
        [
          1,
          384,
          1,
          120,
          52
        ]
      ],
      "ambiguous_among_saved_inputs": false,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_8601bc3a88374d99.pt",
          "input_sha256": "c7f10fe8b8ca84229e785645c70c003456c3950c79cdd176694e5bc86c9247f4",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            384,
            1,
            120,
            52
          ]
        }
      ],
      "model_physical_groups": 0,
      "model_steady_calls": 0,
      "model_clipped_summed_gpu_ms": 0,
      "model_first_event_ids": [],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "void cudnn::engines_precompiled::nchwToNhwcKernel<float, float, float, false, true, (cudnnKernelDataType_t)0>(cudnn::engines_precompiled::nchw2nhwc_params_t<float>, float const*, float*)",
      "grid": [
        2405,
        6,
        1
      ],
      "block": [
        256,
        1,
        1
      ],
      "registers_per_thread": 32,
      "total_shared_memory_bytes": 4224,
      "saved_input_count": 1,
      "distinct_output_shapes": [
        [
          1,
          192,
          1,
          240,
          104
        ]
      ],
      "ambiguous_among_saved_inputs": false,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_99475e040d65d084.pt",
          "input_sha256": "0cc0a21ce474602c53c3e5621e64598ed7be5a2c16603fc968441df41de71330",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            192,
            1,
            240,
            104
          ]
        }
      ],
      "model_physical_groups": 0,
      "model_steady_calls": 0,
      "model_clipped_summed_gpu_ms": 0,
      "model_first_event_ids": [],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "void cudnn::engines_precompiled::nchwToNhwcKernel<float, float, float, false, true, (cudnnKernelDataType_t)0>(cudnn::engines_precompiled::nchw2nhwc_params_t<float>, float const*, float*)",
      "grid": [
        1,
        6,
        192
      ],
      "block": [
        256,
        1,
        1
      ],
      "registers_per_thread": 32,
      "total_shared_memory_bytes": 4224,
      "saved_input_count": 1,
      "distinct_output_shapes": [
        [
          1,
          192,
          1,
          240,
          104
        ]
      ],
      "ambiguous_among_saved_inputs": false,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_99475e040d65d084.pt",
          "input_sha256": "0cc0a21ce474602c53c3e5621e64598ed7be5a2c16603fc968441df41de71330",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            192,
            1,
            240,
            104
          ]
        }
      ],
      "model_physical_groups": 0,
      "model_steady_calls": 0,
      "model_clipped_summed_gpu_ms": 0,
      "model_first_event_ids": [],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize128x128x32_warpgroupsize1x1x1_g1_execute_segment_k_off_kernel__5x_cudnn",
      "grid": [
        132,
        1,
        1
      ],
      "block": [
        384,
        1,
        1
      ],
      "registers_per_thread": 168,
      "total_shared_memory_bytes": 231424,
      "saved_input_count": 1,
      "distinct_output_shapes": [
        [
          1,
          192,
          1,
          240,
          104
        ]
      ],
      "ambiguous_among_saved_inputs": false,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_99475e040d65d084.pt",
          "input_sha256": "0cc0a21ce474602c53c3e5621e64598ed7be5a2c16603fc968441df41de71330",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            192,
            1,
            240,
            104
          ]
        }
      ],
      "model_physical_groups": 4,
      "model_steady_calls": 576,
      "model_clipped_summed_gpu_ms": 332.966756,
      "model_first_event_ids": [
        {
          "physical_gpu": "0",
          "first_event_id": "906637"
        },
        {
          "physical_gpu": "3",
          "first_event_id": "906594"
        },
        {
          "physical_gpu": "2",
          "first_event_id": "906636"
        },
        {
          "physical_gpu": "1",
          "first_event_id": "906643"
        }
      ],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "void cudnn::engines_precompiled::nhwcToNchwKernel<float, float, float, true, false, (cudnnKernelDataType_t)0>(cudnn::engines_precompiled::nhwc2nchw_params_t<float>, float const*, float*)",
      "grid": [
        780,
        6,
        1
      ],
      "block": [
        256,
        1,
        1
      ],
      "registers_per_thread": 32,
      "total_shared_memory_bytes": 4224,
      "saved_input_count": 1,
      "distinct_output_shapes": [
        [
          1,
          192,
          1,
          240,
          104
        ]
      ],
      "ambiguous_among_saved_inputs": false,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_99475e040d65d084.pt",
          "input_sha256": "0cc0a21ce474602c53c3e5621e64598ed7be5a2c16603fc968441df41de71330",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            192,
            1,
            240,
            104
          ]
        }
      ],
      "model_physical_groups": 0,
      "model_steady_calls": 0,
      "model_clipped_summed_gpu_ms": 0,
      "model_first_event_ids": [],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "void at::native::elementwise_kernel<128, 2, at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<float> >(at::TensorIteratorBase&, at::native::CUDAFunctor_add<float> const&)::{lambda(int)#1}>(int, at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<float> >(at::TensorIteratorBase&, at::native::CUDAFunctor_add<float> const&)::{lambda(int)#1})",
      "grid": [
        18720,
        1,
        1
      ],
      "block": [
        128,
        1,
        1
      ],
      "registers_per_thread": 18,
      "total_shared_memory_bytes": 0,
      "saved_input_count": 1,
      "distinct_output_shapes": [
        [
          1,
          192,
          1,
          240,
          104
        ]
      ],
      "ambiguous_among_saved_inputs": false,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_99475e040d65d084.pt",
          "input_sha256": "0cc0a21ce474602c53c3e5621e64598ed7be5a2c16603fc968441df41de71330",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            192,
            1,
            240,
            104
          ]
        }
      ],
      "model_physical_groups": 0,
      "model_steady_calls": 0,
      "model_clipped_summed_gpu_ms": 0,
      "model_first_event_ids": [],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "void cudnn::engines_precompiled::nchwToNhwcKernel<float, float, float, false, true, (cudnnKernelDataType_t)0>(cudnn::engines_precompiled::nchw2nhwc_params_t<float>, float const*, float*)",
      "grid": [
        9490,
        3,
        1
      ],
      "block": [
        256,
        1,
        1
      ],
      "registers_per_thread": 32,
      "total_shared_memory_bytes": 4224,
      "saved_input_count": 1,
      "distinct_output_shapes": [
        [
          1,
          96,
          1,
          480,
          208
        ]
      ],
      "ambiguous_among_saved_inputs": false,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_0bd1bd0a5c44ff3b.pt",
          "input_sha256": "1e47965836d719f164296b5c8d52cac8c22a10d9ff8c5f4fdfa48498909f8856",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            96,
            1,
            480,
            208
          ]
        }
      ],
      "model_physical_groups": 0,
      "model_steady_calls": 0,
      "model_clipped_summed_gpu_ms": 0,
      "model_first_event_ids": [],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "void cudnn::engines_precompiled::nchwToNhwcKernel<float, float, float, false, true, (cudnnKernelDataType_t)0>(cudnn::engines_precompiled::nchw2nhwc_params_t<float>, float const*, float*)",
      "grid": [
        1,
        3,
        96
      ],
      "block": [
        256,
        1,
        1
      ],
      "registers_per_thread": 32,
      "total_shared_memory_bytes": 4224,
      "saved_input_count": 1,
      "distinct_output_shapes": [
        [
          1,
          96,
          1,
          480,
          208
        ]
      ],
      "ambiguous_among_saved_inputs": false,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_0bd1bd0a5c44ff3b.pt",
          "input_sha256": "1e47965836d719f164296b5c8d52cac8c22a10d9ff8c5f4fdfa48498909f8856",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            96,
            1,
            480,
            208
          ]
        }
      ],
      "model_physical_groups": 0,
      "model_steady_calls": 0,
      "model_clipped_summed_gpu_ms": 0,
      "model_first_event_ids": [],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize256x32x32_warpgroupsize1x1x1_g1_execute_segment_k_off_kernel__5x_cudnn",
      "grid": [
        66,
        2,
        1
      ],
      "block": [
        384,
        1,
        1
      ],
      "registers_per_thread": 168,
      "total_shared_memory_bytes": 231424,
      "saved_input_count": 1,
      "distinct_output_shapes": [
        [
          1,
          96,
          1,
          480,
          208
        ]
      ],
      "ambiguous_among_saved_inputs": false,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_0bd1bd0a5c44ff3b.pt",
          "input_sha256": "1e47965836d719f164296b5c8d52cac8c22a10d9ff8c5f4fdfa48498909f8856",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            96,
            1,
            480,
            208
          ]
        }
      ],
      "model_physical_groups": 4,
      "model_steady_calls": 288,
      "model_clipped_summed_gpu_ms": 383.406434,
      "model_first_event_ids": [
        {
          "physical_gpu": "1",
          "first_event_id": "907873"
        },
        {
          "physical_gpu": "0",
          "first_event_id": "907875"
        },
        {
          "physical_gpu": "2",
          "first_event_id": "907874"
        },
        {
          "physical_gpu": "3",
          "first_event_id": "907863"
        }
      ],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "void cudnn::engines_precompiled::nhwcToNchwKernel<float, float, float, true, false, (cudnnKernelDataType_t)0>(cudnn::engines_precompiled::nhwc2nchw_params_t<float>, float const*, float*)",
      "grid": [
        3120,
        3,
        1
      ],
      "block": [
        256,
        1,
        1
      ],
      "registers_per_thread": 32,
      "total_shared_memory_bytes": 4224,
      "saved_input_count": 1,
      "distinct_output_shapes": [
        [
          1,
          96,
          1,
          480,
          208
        ]
      ],
      "ambiguous_among_saved_inputs": false,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_0bd1bd0a5c44ff3b.pt",
          "input_sha256": "1e47965836d719f164296b5c8d52cac8c22a10d9ff8c5f4fdfa48498909f8856",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            96,
            1,
            480,
            208
          ]
        }
      ],
      "model_physical_groups": 0,
      "model_steady_calls": 0,
      "model_clipped_summed_gpu_ms": 0,
      "model_first_event_ids": [],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "void at::native::elementwise_kernel<128, 2, at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<float> >(at::TensorIteratorBase&, at::native::CUDAFunctor_add<float> const&)::{lambda(int)#1}>(int, at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<float> >(at::TensorIteratorBase&, at::native::CUDAFunctor_add<float> const&)::{lambda(int)#1})",
      "grid": [
        37440,
        1,
        1
      ],
      "block": [
        128,
        1,
        1
      ],
      "registers_per_thread": 18,
      "total_shared_memory_bytes": 0,
      "saved_input_count": 1,
      "distinct_output_shapes": [
        [
          1,
          96,
          1,
          480,
          208
        ]
      ],
      "ambiguous_among_saved_inputs": false,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/vae_rank0_0bd1bd0a5c44ff3b.pt",
          "input_sha256": "1e47965836d719f164296b5c8d52cac8c22a10d9ff8c5f4fdfa48498909f8856",
          "op": "aten.conv3d.default",
          "stage": "vae",
          "output_shape": [
            1,
            96,
            1,
            480,
            208
          ]
        }
      ],
      "model_physical_groups": 0,
      "model_steady_calls": 0,
      "model_clipped_summed_gpu_ms": 0,
      "model_first_event_ids": [],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda()#3}::operator()() const::{lambda()#1}::operator()() const::{lambda(unsigned char)#1}, std::array<char*, 2ul>, 4, TrivialOffsetCalculator<1, unsigned int>, TrivialOffsetCalculator<1, unsigned int>, at::native::memory::LoadWithCast<1>, at::native::memory::StoreWithCast<1> >(int, at::native::direct_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda()#3}::operator()() const::{lambda()#1}::operator()() const::{lambda(unsigned char)#1}, std::array<char*, 2ul>, TrivialOffsetCalculator<1, unsigned int>, TrivialOffsetCalculator<1, unsigned int>, at::native::memory::LoadWithCast<1>, at::native::memory::StoreWithCast<1>)",
      "grid": [
        21060,
        1,
        1
      ],
      "block": [
        128,
        1,
        1
      ],
      "registers_per_thread": 32,
      "total_shared_memory_bytes": 0,
      "saved_input_count": 1,
      "distinct_output_shapes": [
        [
          1,
          3,
          9,
          480,
          832
        ]
      ],
      "ambiguous_among_saved_inputs": false,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/output_rank0_f18eb91117dda1c8.pt",
          "input_sha256": "897dd2bd3ad5fe7258ee5c3ae9cfd93da4e5e94644902267f0b3d8bd156f63b0",
          "op": "aten.to.dtype",
          "stage": "output",
          "output_shape": [
            1,
            3,
            9,
            480,
            832
          ]
        }
      ],
      "model_physical_groups": 0,
      "model_steady_calls": 0,
      "model_clipped_summed_gpu_ms": 0,
      "model_first_event_ids": [],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    },
    {
      "name": "void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda()#3}::operator()() const::{lambda()#1}::operator()() const::{lambda(unsigned char)#1}, std::array<char*, 2ul>, 4, TrivialOffsetCalculator<1, unsigned int>, TrivialOffsetCalculator<1, unsigned int>, at::native::memory::LoadWithCast<1>, at::native::memory::StoreWithCast<1> >(int, at::native::direct_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda()#3}::operator()() const::{lambda()#1}::operator()() const::{lambda(unsigned char)#1}, std::array<char*, 2ul>, TrivialOffsetCalculator<1, unsigned int>, TrivialOffsetCalculator<1, unsigned int>, at::native::memory::LoadWithCast<1>, at::native::memory::StoreWithCast<1>)",
      "grid": [
        28080,
        1,
        1
      ],
      "block": [
        128,
        1,
        1
      ],
      "registers_per_thread": 32,
      "total_shared_memory_bytes": 0,
      "saved_input_count": 1,
      "distinct_output_shapes": [
        [
          1,
          3,
          12,
          480,
          832
        ]
      ],
      "ambiguous_among_saved_inputs": false,
      "inputs": [
        {
          "input": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/operator_capture_480x832_v2/operators/output_rank0_e753c57113db57aa.pt",
          "input_sha256": "a9aab61b63e7cab483e81582c9af41d89d525814ab14de753e8a7b16de6c4f1b",
          "op": "aten.to.dtype",
          "stage": "output",
          "output_shape": [
            1,
            3,
            12,
            480,
            832
          ]
        }
      ],
      "model_physical_groups": 0,
      "model_steady_calls": 0,
      "model_clipped_summed_gpu_ms": 0,
      "model_first_event_ids": [],
      "shape_assignment": "unresolved; signatures alone do not identify input shapes"
    }
  ]
}
