{
  "status": "passed_local_comparison_not_model",
  "generated_at": "2026-09-10T04:46:54.448770+00:00",
  "results": {
    "original": {
      "exact_to_saved_output": true,
      "samples": 900,
      "independent_launches": 1,
      "repeat_mean_ms": [
        0.051392426652212934,
        0.05497493349015713,
        0.051326613227526345
      ],
      "mean_of_repeat_means_ms": 0.05256465778996547
    },
    "channels_last_inclusive": {
      "exact_to_saved_output": true,
      "samples": 900,
      "independent_launches": 1,
      "repeat_mean_ms": [
        0.0605838933835427,
        0.06017194665968418,
        0.06030677328507106
      ],
      "mean_of_repeat_means_ms": 0.06035420444276598
    }
  },
  "candidate_time_ratio": 1.1481898100416725,
  "input_sha256": "03221b37d95f7c65a91055eae771f796fbf8e0d7758ce6840701ea0c066358ea",
  "allocated": "6",
  "scope": "One process, same real input and NCDHW output, three interleaved 300-sample batches per mode. All packing/conversion costs included. Local eager CUDA-event timing, no profiler. Samples correlated, not three independent launches or a model P99/stability study. No integrated speedup accepted; a negative layout candidate is not evidence that all VAE layout optimization is impossible.",
  "sha256": {
    "script.py": "9cdc2bf7b264c56fc1d9103ae2a2b665bb454d5b1740a22b8b3ebb710da8d864",
    "original_output.pt": "c8b3e452906c208fa39f90e58c4dd0ed6751927e82f23846a216b369d13ce876",
    "contract.json": "f174fc263b20c7d908ba0a18d6ae1fbec34f1d0cbef227386278945f483934a4",
    "channels_last_inclusive_output.pt": "7f38cf6c678ae49cac5fe9a24f7dd66aad8f666dcfe21a8a53ac8da2ddff6467",
    "gpu_environment.txt": "2f2e6958966ae689b3393ae56b19d04d16e1f6186d59989eccdfe63aaa322a58",
    "manifest.json": "db5c408c37963de7143f876573bb2832bbc88fc4024750772a3752471522148a"
  }
}
