{
  "generated_at": "2026-09-10T02:15:18.599596+00:00",
  "status": "counter_and_output_checks_passed_scheduler_note",
  "scheduler_history": [
    {
      "gpu_id": 6,
      "pid": 613927,
      "user": "unknown",
      "kind": "foreign",
      "holder": "zhoutaichang",
      "holder_account": "zhoutaichang",
      "command": "[No data]",
      "memory_mb": 1254,
      "first_seen": "2026-09-10T02:13:28.673321975Z",
      "last_seen": "2026-09-10T02:13:28.941159451Z",
      "sightings": 3,
      "warn_count": 0,
      "last_warned": null,
      "last_signal": null,
      "end_time": "2026-09-10T02:13:31.091814747Z",
      "resolution": "process exited"
    }
  ],
  "scheduler_note": "Guard briefly classified this PID as unknown user near successful completion, with no warning or signal and resolution process exited. Identity file proves UID2005 at startup. Teardown race is a possible explanation, not proven. Do not claim zero guard records.",
  "jobs": [
    {
      "stage": "vae",
      "status": "passed",
      "output": "/data/zhoutaichang/feature/Lingbot_world_realtime/runs/analysis_report_20260908/raw/ncu_capability_vae_detailed_v1",
      "elapsed_seconds": 25.631391525268555,
      "allocated": "6",
      "input_sha256": "1e47965836d719f164296b5c8d52cac8c22a10d9ff8c5f4fdfa48498909f8856",
      "correctness": {
        "op": "aten.conv3d.default",
        "stage": "vae",
        "shape": [
          1,
          96,
          1,
          480,
          208
        ],
        "exact": true,
        "max_abs": 0.0,
        "rmse": 0.0,
        "finite": true,
        "math_policy": {
          "matmul_precision": "highest",
          "matmul_allow_tf32": false,
          "cudnn_allow_tf32": true,
          "cudnn_benchmark": false,
          "cudnn_deterministic": false
        },
        "scope": "Original ATen op on saved real tensors; isolated diagnostic"
      },
      "kernel_count": 5,
      "kernels": [
        {
          "id": "0",
          "name": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
          "device": "0",
          "metrics": {
            "gpu__time_duration.sum": {
              "value": 69.952,
              "unit": "us"
            },
            "sm__throughput.avg.pct_of_peak_sustained_elapsed": {
              "value": 37.664733,
              "unit": "%"
            },
            "gpu__dram_throughput.avg.pct_of_peak_sustained_elapsed": {
              "value": 64.014557,
              "unit": "%"
            },
            "lts__throughput.avg.pct_of_peak_sustained_elapsed": {
              "value": 81.220531,
              "unit": "%"
            },
            "sm__warps_active.avg.pct_of_peak_sustained_active": {
              "value": 92.804217,
              "unit": "%"
            },
            "dram__bytes_read.sum": {
              "value": 116.621568,
              "unit": "Mbyte"
            },
            "dram__bytes_write.sum": {
              "value": 98.635264,
              "unit": "Mbyte"
            },
            "dram__bytes.sum.per_second": {
              "value": 3.077208,
              "unit": "Tbyte/s"
            },
            "launch__registers_per_thread": {
              "value": 32.0,
              "unit": "register/thread"
            },
            "launch__shared_mem_per_block": {
              "value": 5.248,
              "unit": "Kbyte/block"
            }
          }
        },
        {
          "id": "1",
          "name": "void cudnn::nchwToNhwcKernel<float, float, float, 0, 1, 0>(cudnn::nchw2nhwc_params_t<T3>, const T1 *, T2 *)",
          "device": "0",
          "metrics": {
            "gpu__time_duration.sum": {
              "value": 3.52,
              "unit": "us"
            },
            "sm__throughput.avg.pct_of_peak_sustained_elapsed": {
              "value": 12.364265,
              "unit": "%"
            },
            "gpu__dram_throughput.avg.pct_of_peak_sustained_elapsed": {
              "value": 5.962006,
              "unit": "%"
            },
            "lts__throughput.avg.pct_of_peak_sustained_elapsed": {
              "value": 15.411113,
              "unit": "%"
            },
            "sm__warps_active.avg.pct_of_peak_sustained_active": {
              "value": 24.673274,
              "unit": "%"
            },
            "dram__bytes_read.sum": {
              "value": 1.004288,
              "unit": "Mbyte"
            },
            "dram__bytes_write.sum": {
              "value": 0.0,
              "unit": "Mbyte"
            },
            "dram__bytes.sum.per_second": {
              "value": 0.285309,
              "unit": "Tbyte/s"
            },
            "launch__registers_per_thread": {
              "value": 32.0,
              "unit": "register/thread"
            },
            "launch__shared_mem_per_block": {
              "value": 5.248,
              "unit": "Kbyte/block"
            }
          }
        },
        {
          "id": "2",
          "name": "sm90_xmma_fprop_implicit_gemm_f32f32_tf32f32_f32_nhwckrsc_nhwc_tilesize256x32x32_warpgroupsize1x1x1_g1_execute_segment_k_off_kernel__5x_cudnn",
          "device": "0",
          "metrics": {
            "gpu__time_duration.sum": {
              "value": 343.968,
              "unit": "us"
            },
            "sm__throughput.avg.pct_of_peak_sustained_elapsed": {
              "value": 29.783099,
              "unit": "%"
            },
            "gpu__dram_throughput.avg.pct_of_peak_sustained_elapsed": {
              "value": 9.053182,
              "unit": "%"
            },
            "lts__throughput.avg.pct_of_peak_sustained_elapsed": {
              "value": 60.36853,
              "unit": "%"
            },
            "sm__warps_active.avg.pct_of_peak_sustained_active": {
              "value": 17.59726,
              "unit": "%"
            },
            "dram__bytes_read.sum": {
              "value": 117.662464,
              "unit": "Mbyte"
            },
            "dram__bytes_write.sum": {
              "value": 32.175104,
              "unit": "Mbyte"
            },
            "dram__bytes.sum.per_second": {
              "value": 0.435615,
              "unit": "Tbyte/s"
            },
            "launch__registers_per_thread": {
              "value": 168.0,
              "unit": "register/thread"
            },
            "launch__shared_mem_per_block": {
              "value": 232.448,
              "unit": "Kbyte/block"
            }
          }
        },
        {
          "id": "3",
          "name": "void cudnn::nhwcToNchwKernel<float, float, float, 1, 0, 0>(cudnn::nhwc2nchw_params_t<T3>, const T1 *, T2 *)",
          "device": "0",
          "metrics": {
            "gpu__time_duration.sum": {
              "value": 19.968,
              "unit": "us"
            },
            "sm__throughput.avg.pct_of_peak_sustained_elapsed": {
              "value": 41.449384,
              "unit": "%"
            },
            "gpu__dram_throughput.avg.pct_of_peak_sustained_elapsed": {
              "value": 59.726043,
              "unit": "%"
            },
            "lts__throughput.avg.pct_of_peak_sustained_elapsed": {
              "value": 81.746478,
              "unit": "%"
            },
            "sm__warps_active.avg.pct_of_peak_sustained_active": {
              "value": 88.769719,
              "unit": "%"
            },
            "dram__bytes_read.sum": {
              "value": 38.35648,
              "unit": "Mbyte"
            },
            "dram__bytes_write.sum": {
              "value": 18.939392,
              "unit": "Mbyte"
            },
            "dram__bytes.sum.per_second": {
              "value": 2.869385,
              "unit": "Tbyte/s"
            },
            "launch__registers_per_thread": {
              "value": 32.0,
              "unit": "register/thread"
            },
            "launch__shared_mem_per_block": {
              "value": 5.248,
              "unit": "Kbyte/block"
            }
          }
        },
        {
          "id": "4",
          "name": "void at::elementwise_kernel<128, 2, void at::gpu_kernel_impl_nocast<at::CUDAFunctor_add<float>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)",
          "device": "0",
          "metrics": {
            "gpu__time_duration.sum": {
              "value": 27.616,
              "unit": "us"
            },
            "sm__throughput.avg.pct_of_peak_sustained_elapsed": {
              "value": 58.140107,
              "unit": "%"
            },
            "gpu__dram_throughput.avg.pct_of_peak_sustained_elapsed": {
              "value": 38.876093,
              "unit": "%"
            },
            "lts__throughput.avg.pct_of_peak_sustained_elapsed": {
              "value": 59.812679,
              "unit": "%"
            },
            "sm__warps_active.avg.pct_of_peak_sustained_active": {
              "value": 79.875376,
              "unit": "%"
            },
            "dram__bytes_read.sum": {
              "value": 38.354432,
              "unit": "Mbyte"
            },
            "dram__bytes_write.sum": {
              "value": 13.28128,
              "unit": "Mbyte"
            },
            "dram__bytes.sum.per_second": {
              "value": 1.869775,
              "unit": "Tbyte/s"
            },
            "launch__registers_per_thread": {
              "value": 18.0,
              "unit": "register/thread"
            },
            "launch__shared_mem_per_block": {
              "value": 1.024,
              "unit": "Kbyte/block"
            }
          }
        }
      ],
      "sha256": {
        "replay_comparison.json": "fbc0f1cc0275724c75a7685d0c3da84472740f5dee6ac7697f4160715261d45b",
        "metrics.csv": "a941df9d7f493f0e835a5bd5568ef2ba7ef956d1eda8f84287bcfa85bd03c611",
        "operator.ncu-rep": "caae1bd2de23c7f57ba2022f0364fc6a7d0825b16c2d360467c6f3052e24a7c0",
        "run.log": "97ab50cedc3069a4be10e51e9402c935c8e7d4d4a233e8ab220f73b3821be506",
        "identity.json": "d40f8c3e3e337af1267261b76dcbfc5f969ad3ca723ad289c8c724aa8052f955",
        "replay_operator.py": "21e6ebbe13e53933591ad14feaa6ab4b189738f3647f2087ce8bf523b0e07450",
        "manifest.json": "bc35fb4b0f3632e57056326aaf8ff5f5498bda0f6f6bbe2f90738bdd78073842",
        "guard_own_pid_history_unfiltered.json": "1dc4fba81c2119cd2b305d17d9a524473c2acd16c59a81f484ec3ec07cace098",
        "guard_own_pid_history.json": "2ae473cf5fdff3494bc9b9c7c7e8a537824fae83f2b2a011c7a4303c43732e96"
      }
    }
  ],
  "scope": "One VAE real-input operator replay with detailed scheduler/warp/source counters, single device per job and distinct allocated devices across jobs. Exact reference output passed. NCU replay instrumentation and clocks/cache effects prevent treating these timings as clean integrated performance. No whole-model MFU or optimization acceptance. Guard history is a point-in-time check of each own PID."
}
