{
  "schema_version": 1,
  "verified_at_utc": "2026-09-15T00:00:50.565257+00:00",
  "analyzer_sha256": "fc34c05578da57c94105165cf8ee5af9abc335d18ad4f27221364c4771c1e6f9",
  "base_protocol_sha256": "ec2c0c3cb8037f11766604c90c1cd7352e8656bea59d4d3ab3da2c7e1ce75496",
  "plan_sha256": "a42d71925d706406e6089c6c10047ba7dca99a710bf4d38f15404038b9af5873",
  "frozen_base_sources_unchanged": true,
  "status": "complete",
  "all_available_evidence_verified": true,
  "all_jobs_passed": true,
  "pending_jobs": 0,
  "failed_or_unmeasured_jobs": 0,
  "qualification": {
    "complete": true,
    "draws_bit_equal": 11,
    "predictions_bit_equal": 110000,
    "draws": [
      {
        "result_path": "qualify-00.json",
        "result_sha256": "172f9a0079f3b28ecea67767ddd5de877574ae7a22b071b6fd0c0b31c19cfdc9",
        "result_uncompressed_sha256": "172f9a0079f3b28ecea67767ddd5de877574ae7a22b071b6fd0c0b31c19cfdc9",
        "result_uncompressed_bytes": 26643,
        "result_compressed_sha256": null,
        "result_compressed_bytes": null,
        "result_gzip_contents_verified": false,
        "command": [
          "/usr/local/bin/python",
          "/workspace/cache/fixture.py",
          "--input",
          "/tmp/cache-run/input.npz",
          "--config",
          "/tmp/cache-run/config.json",
          "--output",
          "/tmp/cache-run/result.json",
          "--mode",
          "measure",
          "--scope",
          "task",
          "--repeats",
          "16"
        ],
        "capabilities": {
          "ncu_version": "NVIDIA (R) Nsight Compute Command Line Profiler\nCopyright (c) 2018-2025 NVIDIA Corporation\nVersion 2025.1.1.0 (build 35528883) (public-release)\n"
        },
        "returncode": 0,
        "workspace": ":16:8",
        "scope": "task",
        "status": "passed",
        "draw": 0,
        "predictions_bit_equal": true,
        "all_state_hashes_bit_equal": false,
        "state_fields_differing_from_original": [
          "final_parameter_sha256",
          "final_velocity_sha256",
          "state_vector_sha256",
          "scores_sha256"
        ],
        "state_fields_checked": [
          "input_hashes",
          "initial_parameter_sha256",
          "epoch_permutation_sha256",
          "final_parameter_sha256",
          "final_velocity_sha256",
          "state_vector_sha256",
          "predictions_sha256",
          "scores_sha256"
        ],
        "prediction_path": "qualify-00-result.predictions.npy",
        "prediction_sha256": "fce8a9458d2158537c7d6207fb0c8e9cfcbba007e7a748976bf3e7cd002729f3",
        "memory": {
          "peak_allocated_bytes": 15663104,
          "live_allocated_bytes": 14237696,
          "peak_reserved_bytes": 29360128,
          "named_storage_bytes": 13960864,
          "unnamed_live_allocated_bytes": 276832,
          "peak_allocated_mib": 14.9375,
          "nominal_l2_reference_bytes": 41943040,
          "peak_allocated_below_nominal_l2_capacity": true,
          "cache_residency_established": false,
          "scope": "Fresh task preparation including CUDA graph capture plus one complete reset/train/predict invocation; tensor allocator only, excluding driver/context. Recorded before output diagnostics and profiling warmup."
        },
        "software": {
          "python": "3.11.15",
          "torch": "2.5.1+cu124",
          "cuda": "12.4",
          "numpy": "2.2.6"
        }
      },
      {
        "result_path": "qualify-01.json",
        "result_sha256": "c6eb56a4d56c4d97755d11ca2b49a72997f6f0cf91799e881f641121fa6e0924",
        "result_uncompressed_sha256": "c6eb56a4d56c4d97755d11ca2b49a72997f6f0cf91799e881f641121fa6e0924",
        "result_uncompressed_bytes": 26641,
        "result_compressed_sha256": null,
        "result_compressed_bytes": null,
        "result_gzip_contents_verified": false,
        "command": [
          "/usr/local/bin/python",
          "/workspace/cache/fixture.py",
          "--input",
          "/tmp/cache-run/input.npz",
          "--config",
          "/tmp/cache-run/config.json",
          "--output",
          "/tmp/cache-run/result.json",
          "--mode",
          "measure",
          "--scope",
          "task",
          "--repeats",
          "16"
        ],
        "capabilities": {
          "ncu_version": "NVIDIA (R) Nsight Compute Command Line Profiler\nCopyright (c) 2018-2025 NVIDIA Corporation\nVersion 2025.1.1.0 (build 35528883) (public-release)\n"
        },
        "returncode": 0,
        "workspace": ":16:8",
        "scope": "task",
        "status": "passed",
        "draw": 1,
        "predictions_bit_equal": true,
        "all_state_hashes_bit_equal": false,
        "state_fields_differing_from_original": [
          "final_parameter_sha256",
          "final_velocity_sha256",
          "state_vector_sha256",
          "scores_sha256"
        ],
        "state_fields_checked": [
          "input_hashes",
          "initial_parameter_sha256",
          "epoch_permutation_sha256",
          "final_parameter_sha256",
          "final_velocity_sha256",
          "state_vector_sha256",
          "predictions_sha256",
          "scores_sha256"
        ],
        "prediction_path": "qualify-01-result.predictions.npy",
        "prediction_sha256": "dd55eb6ae12c65660402dbd4a9de852e79682687a686b4d38c94a41e49ce6b1a",
        "memory": {
          "peak_allocated_bytes": 15663104,
          "live_allocated_bytes": 14237696,
          "peak_reserved_bytes": 29360128,
          "named_storage_bytes": 13960864,
          "unnamed_live_allocated_bytes": 276832,
          "peak_allocated_mib": 14.9375,
          "nominal_l2_reference_bytes": 41943040,
          "peak_allocated_below_nominal_l2_capacity": true,
          "cache_residency_established": false,
          "scope": "Fresh task preparation including CUDA graph capture plus one complete reset/train/predict invocation; tensor allocator only, excluding driver/context. Recorded before output diagnostics and profiling warmup."
        },
        "software": {
          "python": "3.11.15",
          "torch": "2.5.1+cu124",
          "cuda": "12.4",
          "numpy": "2.2.6"
        }
      },
      {
        "result_path": "qualify-02.json",
        "result_sha256": "ce45fd2d498af8dad7d8b0cbaf8677ef1bf13c9f6413759a3c5c42b525a2cb8d",
        "result_uncompressed_sha256": "ce45fd2d498af8dad7d8b0cbaf8677ef1bf13c9f6413759a3c5c42b525a2cb8d",
        "result_uncompressed_bytes": 26643,
        "result_compressed_sha256": null,
        "result_compressed_bytes": null,
        "result_gzip_contents_verified": false,
        "command": [
          "/usr/local/bin/python",
          "/workspace/cache/fixture.py",
          "--input",
          "/tmp/cache-run/input.npz",
          "--config",
          "/tmp/cache-run/config.json",
          "--output",
          "/tmp/cache-run/result.json",
          "--mode",
          "measure",
          "--scope",
          "task",
          "--repeats",
          "16"
        ],
        "capabilities": {
          "ncu_version": "NVIDIA (R) Nsight Compute Command Line Profiler\nCopyright (c) 2018-2025 NVIDIA Corporation\nVersion 2025.1.1.0 (build 35528883) (public-release)\n"
        },
        "returncode": 0,
        "workspace": ":16:8",
        "scope": "task",
        "status": "passed",
        "draw": 2,
        "predictions_bit_equal": true,
        "all_state_hashes_bit_equal": false,
        "state_fields_differing_from_original": [
          "final_parameter_sha256",
          "final_velocity_sha256",
          "state_vector_sha256",
          "scores_sha256"
        ],
        "state_fields_checked": [
          "input_hashes",
          "initial_parameter_sha256",
          "epoch_permutation_sha256",
          "final_parameter_sha256",
          "final_velocity_sha256",
          "state_vector_sha256",
          "predictions_sha256",
          "scores_sha256"
        ],
        "prediction_path": "qualify-02-result.predictions.npy",
        "prediction_sha256": "75435ffba563c4ff588983f065521d6d8a2adcceb9b2ed5125125faa03c9d601",
        "memory": {
          "peak_allocated_bytes": 15663104,
          "live_allocated_bytes": 14237696,
          "peak_reserved_bytes": 29360128,
          "named_storage_bytes": 13960864,
          "unnamed_live_allocated_bytes": 276832,
          "peak_allocated_mib": 14.9375,
          "nominal_l2_reference_bytes": 41943040,
          "peak_allocated_below_nominal_l2_capacity": true,
          "cache_residency_established": false,
          "scope": "Fresh task preparation including CUDA graph capture plus one complete reset/train/predict invocation; tensor allocator only, excluding driver/context. Recorded before output diagnostics and profiling warmup."
        },
        "software": {
          "python": "3.11.15",
          "torch": "2.5.1+cu124",
          "cuda": "12.4",
          "numpy": "2.2.6"
        }
      },
      {
        "result_path": "qualify-03.json",
        "result_sha256": "17f73993894804358ea4620a9036fda5dea497e692d0cb65be116e0440e3b2a4",
        "result_uncompressed_sha256": "17f73993894804358ea4620a9036fda5dea497e692d0cb65be116e0440e3b2a4",
        "result_uncompressed_bytes": 26643,
        "result_compressed_sha256": null,
        "result_compressed_bytes": null,
        "result_gzip_contents_verified": false,
        "command": [
          "/usr/local/bin/python",
          "/workspace/cache/fixture.py",
          "--input",
          "/tmp/cache-run/input.npz",
          "--config",
          "/tmp/cache-run/config.json",
          "--output",
          "/tmp/cache-run/result.json",
          "--mode",
          "measure",
          "--scope",
          "task",
          "--repeats",
          "16"
        ],
        "capabilities": {
          "ncu_version": "NVIDIA (R) Nsight Compute Command Line Profiler\nCopyright (c) 2018-2025 NVIDIA Corporation\nVersion 2025.1.1.0 (build 35528883) (public-release)\n"
        },
        "returncode": 0,
        "workspace": ":16:8",
        "scope": "task",
        "status": "passed",
        "draw": 3,
        "predictions_bit_equal": true,
        "all_state_hashes_bit_equal": false,
        "state_fields_differing_from_original": [
          "final_parameter_sha256",
          "final_velocity_sha256",
          "state_vector_sha256",
          "scores_sha256"
        ],
        "state_fields_checked": [
          "input_hashes",
          "initial_parameter_sha256",
          "epoch_permutation_sha256",
          "final_parameter_sha256",
          "final_velocity_sha256",
          "state_vector_sha256",
          "predictions_sha256",
          "scores_sha256"
        ],
        "prediction_path": "qualify-03-result.predictions.npy",
        "prediction_sha256": "393294bf08f96c82f6ab6e5c88442d9cb6329ca7274b01eed95ec0bcf1ef1fff",
        "memory": {
          "peak_allocated_bytes": 15663104,
          "live_allocated_bytes": 14237696,
          "peak_reserved_bytes": 29360128,
          "named_storage_bytes": 13960864,
          "unnamed_live_allocated_bytes": 276832,
          "peak_allocated_mib": 14.9375,
          "nominal_l2_reference_bytes": 41943040,
          "peak_allocated_below_nominal_l2_capacity": true,
          "cache_residency_established": false,
          "scope": "Fresh task preparation including CUDA graph capture plus one complete reset/train/predict invocation; tensor allocator only, excluding driver/context. Recorded before output diagnostics and profiling warmup."
        },
        "software": {
          "python": "3.11.15",
          "torch": "2.5.1+cu124",
          "cuda": "12.4",
          "numpy": "2.2.6"
        }
      },
      {
        "result_path": "qualify-04.json",
        "result_sha256": "bcbf4958459ff4a0abcaf092aad2060fe01da05c2410546317f17857e4220c98",
        "result_uncompressed_sha256": "bcbf4958459ff4a0abcaf092aad2060fe01da05c2410546317f17857e4220c98",
        "result_uncompressed_bytes": 26642,
        "result_compressed_sha256": null,
        "result_compressed_bytes": null,
        "result_gzip_contents_verified": false,
        "command": [
          "/usr/local/bin/python",
          "/workspace/cache/fixture.py",
          "--input",
          "/tmp/cache-run/input.npz",
          "--config",
          "/tmp/cache-run/config.json",
          "--output",
          "/tmp/cache-run/result.json",
          "--mode",
          "measure",
          "--scope",
          "task",
          "--repeats",
          "16"
        ],
        "capabilities": {
          "ncu_version": "NVIDIA (R) Nsight Compute Command Line Profiler\nCopyright (c) 2018-2025 NVIDIA Corporation\nVersion 2025.1.1.0 (build 35528883) (public-release)\n"
        },
        "returncode": 0,
        "workspace": ":16:8",
        "scope": "task",
        "status": "passed",
        "draw": 4,
        "predictions_bit_equal": true,
        "all_state_hashes_bit_equal": false,
        "state_fields_differing_from_original": [
          "final_parameter_sha256",
          "final_velocity_sha256",
          "state_vector_sha256",
          "scores_sha256"
        ],
        "state_fields_checked": [
          "input_hashes",
          "initial_parameter_sha256",
          "epoch_permutation_sha256",
          "final_parameter_sha256",
          "final_velocity_sha256",
          "state_vector_sha256",
          "predictions_sha256",
          "scores_sha256"
        ],
        "prediction_path": "qualify-04-result.predictions.npy",
        "prediction_sha256": "5cb4bed52d62677ca958300d802cb91abf03a6c63722ebfb1f9bf67d0d65044c",
        "memory": {
          "peak_allocated_bytes": 15663104,
          "live_allocated_bytes": 14237696,
          "peak_reserved_bytes": 29360128,
          "named_storage_bytes": 13960864,
          "unnamed_live_allocated_bytes": 276832,
          "peak_allocated_mib": 14.9375,
          "nominal_l2_reference_bytes": 41943040,
          "peak_allocated_below_nominal_l2_capacity": true,
          "cache_residency_established": false,
          "scope": "Fresh task preparation including CUDA graph capture plus one complete reset/train/predict invocation; tensor allocator only, excluding driver/context. Recorded before output diagnostics and profiling warmup."
        },
        "software": {
          "python": "3.11.15",
          "torch": "2.5.1+cu124",
          "cuda": "12.4",
          "numpy": "2.2.6"
        }
      },
      {
        "result_path": "qualify-05.json",
        "result_sha256": "a7df892969cf1a1b7cbe4770d11a41c1ccbbc3b39ad21f2935c04ddff8510027",
        "result_uncompressed_sha256": "a7df892969cf1a1b7cbe4770d11a41c1ccbbc3b39ad21f2935c04ddff8510027",
        "result_uncompressed_bytes": 26642,
        "result_compressed_sha256": null,
        "result_compressed_bytes": null,
        "result_gzip_contents_verified": false,
        "command": [
          "/usr/local/bin/python",
          "/workspace/cache/fixture.py",
          "--input",
          "/tmp/cache-run/input.npz",
          "--config",
          "/tmp/cache-run/config.json",
          "--output",
          "/tmp/cache-run/result.json",
          "--mode",
          "measure",
          "--scope",
          "task",
          "--repeats",
          "16"
        ],
        "capabilities": {
          "ncu_version": "NVIDIA (R) Nsight Compute Command Line Profiler\nCopyright (c) 2018-2025 NVIDIA Corporation\nVersion 2025.1.1.0 (build 35528883) (public-release)\n"
        },
        "returncode": 0,
        "workspace": ":16:8",
        "scope": "task",
        "status": "passed",
        "draw": 5,
        "predictions_bit_equal": true,
        "all_state_hashes_bit_equal": false,
        "state_fields_differing_from_original": [
          "final_parameter_sha256",
          "final_velocity_sha256",
          "state_vector_sha256",
          "scores_sha256"
        ],
        "state_fields_checked": [
          "input_hashes",
          "initial_parameter_sha256",
          "epoch_permutation_sha256",
          "final_parameter_sha256",
          "final_velocity_sha256",
          "state_vector_sha256",
          "predictions_sha256",
          "scores_sha256"
        ],
        "prediction_path": "qualify-05-result.predictions.npy",
        "prediction_sha256": "a668e0ec96783c245aaf27bf6edb2170e0efbaf2346851eb30aa737c890d6453",
        "memory": {
          "peak_allocated_bytes": 15663104,
          "live_allocated_bytes": 14237696,
          "peak_reserved_bytes": 29360128,
          "named_storage_bytes": 13960864,
          "unnamed_live_allocated_bytes": 276832,
          "peak_allocated_mib": 14.9375,
          "nominal_l2_reference_bytes": 41943040,
          "peak_allocated_below_nominal_l2_capacity": true,
          "cache_residency_established": false,
          "scope": "Fresh task preparation including CUDA graph capture plus one complete reset/train/predict invocation; tensor allocator only, excluding driver/context. Recorded before output diagnostics and profiling warmup."
        },
        "software": {
          "python": "3.11.15",
          "torch": "2.5.1+cu124",
          "cuda": "12.4",
          "numpy": "2.2.6"
        }
      },
      {
        "result_path": "qualify-06.json",
        "result_sha256": "ceeaf92e6053fcbfb457d7fd55bd60aa5d02385873ae6aaaab0353add503acb8",
        "result_uncompressed_sha256": "ceeaf92e6053fcbfb457d7fd55bd60aa5d02385873ae6aaaab0353add503acb8",
        "result_uncompressed_bytes": 26643,
        "result_compressed_sha256": null,
        "result_compressed_bytes": null,
        "result_gzip_contents_verified": false,
        "command": [
          "/usr/local/bin/python",
          "/workspace/cache/fixture.py",
          "--input",
          "/tmp/cache-run/input.npz",
          "--config",
          "/tmp/cache-run/config.json",
          "--output",
          "/tmp/cache-run/result.json",
          "--mode",
          "measure",
          "--scope",
          "task",
          "--repeats",
          "16"
        ],
        "capabilities": {
          "ncu_version": "NVIDIA (R) Nsight Compute Command Line Profiler\nCopyright (c) 2018-2025 NVIDIA Corporation\nVersion 2025.1.1.0 (build 35528883) (public-release)\n"
        },
        "returncode": 0,
        "workspace": ":16:8",
        "scope": "task",
        "status": "passed",
        "draw": 6,
        "predictions_bit_equal": true,
        "all_state_hashes_bit_equal": false,
        "state_fields_differing_from_original": [
          "final_parameter_sha256",
          "final_velocity_sha256",
          "state_vector_sha256",
          "scores_sha256"
        ],
        "state_fields_checked": [
          "input_hashes",
          "initial_parameter_sha256",
          "epoch_permutation_sha256",
          "final_parameter_sha256",
          "final_velocity_sha256",
          "state_vector_sha256",
          "predictions_sha256",
          "scores_sha256"
        ],
        "prediction_path": "qualify-06-result.predictions.npy",
        "prediction_sha256": "081a7ac2a89ee6181fd8dd40f1d167898b9b6c8f0539ae6b15f4a8edb0f2b8d2",
        "memory": {
          "peak_allocated_bytes": 15663104,
          "live_allocated_bytes": 14237696,
          "peak_reserved_bytes": 29360128,
          "named_storage_bytes": 13960864,
          "unnamed_live_allocated_bytes": 276832,
          "peak_allocated_mib": 14.9375,
          "nominal_l2_reference_bytes": 41943040,
          "peak_allocated_below_nominal_l2_capacity": true,
          "cache_residency_established": false,
          "scope": "Fresh task preparation including CUDA graph capture plus one complete reset/train/predict invocation; tensor allocator only, excluding driver/context. Recorded before output diagnostics and profiling warmup."
        },
        "software": {
          "python": "3.11.15",
          "torch": "2.5.1+cu124",
          "cuda": "12.4",
          "numpy": "2.2.6"
        }
      },
      {
        "result_path": "qualify-07.json",
        "result_sha256": "715ebdf8e07da5b9aa780f55b929f7fa5f028fc807c0686543e57b2d2e5f1b60",
        "result_uncompressed_sha256": "715ebdf8e07da5b9aa780f55b929f7fa5f028fc807c0686543e57b2d2e5f1b60",
        "result_uncompressed_bytes": 26643,
        "result_compressed_sha256": null,
        "result_compressed_bytes": null,
        "result_gzip_contents_verified": false,
        "command": [
          "/usr/local/bin/python",
          "/workspace/cache/fixture.py",
          "--input",
          "/tmp/cache-run/input.npz",
          "--config",
          "/tmp/cache-run/config.json",
          "--output",
          "/tmp/cache-run/result.json",
          "--mode",
          "measure",
          "--scope",
          "task",
          "--repeats",
          "16"
        ],
        "capabilities": {
          "ncu_version": "NVIDIA (R) Nsight Compute Command Line Profiler\nCopyright (c) 2018-2025 NVIDIA Corporation\nVersion 2025.1.1.0 (build 35528883) (public-release)\n"
        },
        "returncode": 0,
        "workspace": ":16:8",
        "scope": "task",
        "status": "passed",
        "draw": 7,
        "predictions_bit_equal": true,
        "all_state_hashes_bit_equal": false,
        "state_fields_differing_from_original": [
          "final_parameter_sha256",
          "final_velocity_sha256",
          "state_vector_sha256",
          "scores_sha256"
        ],
        "state_fields_checked": [
          "input_hashes",
          "initial_parameter_sha256",
          "epoch_permutation_sha256",
          "final_parameter_sha256",
          "final_velocity_sha256",
          "state_vector_sha256",
          "predictions_sha256",
          "scores_sha256"
        ],
        "prediction_path": "qualify-07-result.predictions.npy",
        "prediction_sha256": "30b0ad45eb3220be4cbf88281ebb4b77ef87e79228d7039a309fbf37f47a804a",
        "memory": {
          "peak_allocated_bytes": 15663104,
          "live_allocated_bytes": 14237696,
          "peak_reserved_bytes": 29360128,
          "named_storage_bytes": 13960864,
          "unnamed_live_allocated_bytes": 276832,
          "peak_allocated_mib": 14.9375,
          "nominal_l2_reference_bytes": 41943040,
          "peak_allocated_below_nominal_l2_capacity": true,
          "cache_residency_established": false,
          "scope": "Fresh task preparation including CUDA graph capture plus one complete reset/train/predict invocation; tensor allocator only, excluding driver/context. Recorded before output diagnostics and profiling warmup."
        },
        "software": {
          "python": "3.11.15",
          "torch": "2.5.1+cu124",
          "cuda": "12.4",
          "numpy": "2.2.6"
        }
      },
      {
        "result_path": "qualify-08.json",
        "result_sha256": "83f6cadf3eeba1dbaed45b7fd7fd1e0fd1a6b97dfe19496c2c2a9ad4bfc1458f",
        "result_uncompressed_sha256": "83f6cadf3eeba1dbaed45b7fd7fd1e0fd1a6b97dfe19496c2c2a9ad4bfc1458f",
        "result_uncompressed_bytes": 26642,
        "result_compressed_sha256": null,
        "result_compressed_bytes": null,
        "result_gzip_contents_verified": false,
        "command": [
          "/usr/local/bin/python",
          "/workspace/cache/fixture.py",
          "--input",
          "/tmp/cache-run/input.npz",
          "--config",
          "/tmp/cache-run/config.json",
          "--output",
          "/tmp/cache-run/result.json",
          "--mode",
          "measure",
          "--scope",
          "task",
          "--repeats",
          "16"
        ],
        "capabilities": {
          "ncu_version": "NVIDIA (R) Nsight Compute Command Line Profiler\nCopyright (c) 2018-2025 NVIDIA Corporation\nVersion 2025.1.1.0 (build 35528883) (public-release)\n"
        },
        "returncode": 0,
        "workspace": ":16:8",
        "scope": "task",
        "status": "passed",
        "draw": 8,
        "predictions_bit_equal": true,
        "all_state_hashes_bit_equal": false,
        "state_fields_differing_from_original": [
          "final_parameter_sha256",
          "final_velocity_sha256",
          "state_vector_sha256",
          "scores_sha256"
        ],
        "state_fields_checked": [
          "input_hashes",
          "initial_parameter_sha256",
          "epoch_permutation_sha256",
          "final_parameter_sha256",
          "final_velocity_sha256",
          "state_vector_sha256",
          "predictions_sha256",
          "scores_sha256"
        ],
        "prediction_path": "qualify-08-result.predictions.npy",
        "prediction_sha256": "e58db6fd9957a160c2a340522bfa101a3cd3e798b7e95e7b26bb8819b9ba9d65",
        "memory": {
          "peak_allocated_bytes": 15663104,
          "live_allocated_bytes": 14237696,
          "peak_reserved_bytes": 29360128,
          "named_storage_bytes": 13960864,
          "unnamed_live_allocated_bytes": 276832,
          "peak_allocated_mib": 14.9375,
          "nominal_l2_reference_bytes": 41943040,
          "peak_allocated_below_nominal_l2_capacity": true,
          "cache_residency_established": false,
          "scope": "Fresh task preparation including CUDA graph capture plus one complete reset/train/predict invocation; tensor allocator only, excluding driver/context. Recorded before output diagnostics and profiling warmup."
        },
        "software": {
          "python": "3.11.15",
          "torch": "2.5.1+cu124",
          "cuda": "12.4",
          "numpy": "2.2.6"
        }
      },
      {
        "result_path": "qualify-09.json",
        "result_sha256": "6fb5cf87ff6f534344d464e16509dffe589b09d0590bea1ab035798e29ea150f",
        "result_uncompressed_sha256": "6fb5cf87ff6f534344d464e16509dffe589b09d0590bea1ab035798e29ea150f",
        "result_uncompressed_bytes": 26643,
        "result_compressed_sha256": null,
        "result_compressed_bytes": null,
        "result_gzip_contents_verified": false,
        "command": [
          "/usr/local/bin/python",
          "/workspace/cache/fixture.py",
          "--input",
          "/tmp/cache-run/input.npz",
          "--config",
          "/tmp/cache-run/config.json",
          "--output",
          "/tmp/cache-run/result.json",
          "--mode",
          "measure",
          "--scope",
          "task",
          "--repeats",
          "16"
        ],
        "capabilities": {
          "ncu_version": "NVIDIA (R) Nsight Compute Command Line Profiler\nCopyright (c) 2018-2025 NVIDIA Corporation\nVersion 2025.1.1.0 (build 35528883) (public-release)\n"
        },
        "returncode": 0,
        "workspace": ":16:8",
        "scope": "task",
        "status": "passed",
        "draw": 9,
        "predictions_bit_equal": true,
        "all_state_hashes_bit_equal": false,
        "state_fields_differing_from_original": [
          "final_parameter_sha256",
          "final_velocity_sha256",
          "state_vector_sha256",
          "scores_sha256"
        ],
        "state_fields_checked": [
          "input_hashes",
          "initial_parameter_sha256",
          "epoch_permutation_sha256",
          "final_parameter_sha256",
          "final_velocity_sha256",
          "state_vector_sha256",
          "predictions_sha256",
          "scores_sha256"
        ],
        "prediction_path": "qualify-09-result.predictions.npy",
        "prediction_sha256": "bac8a4cbfd9ae48999f195535e4a5dcc17cd475a4b212d39376b7cf937070fda",
        "memory": {
          "peak_allocated_bytes": 15663104,
          "live_allocated_bytes": 14237696,
          "peak_reserved_bytes": 29360128,
          "named_storage_bytes": 13960864,
          "unnamed_live_allocated_bytes": 276832,
          "peak_allocated_mib": 14.9375,
          "nominal_l2_reference_bytes": 41943040,
          "peak_allocated_below_nominal_l2_capacity": true,
          "cache_residency_established": false,
          "scope": "Fresh task preparation including CUDA graph capture plus one complete reset/train/predict invocation; tensor allocator only, excluding driver/context. Recorded before output diagnostics and profiling warmup."
        },
        "software": {
          "python": "3.11.15",
          "torch": "2.5.1+cu124",
          "cuda": "12.4",
          "numpy": "2.2.6"
        }
      },
      {
        "result_path": "qualify-10.json",
        "result_sha256": "b17ad1ad08074f6f00ca263cf7a820e8f2f0f016c7821d132327b54131f2851f",
        "result_uncompressed_sha256": "b17ad1ad08074f6f00ca263cf7a820e8f2f0f016c7821d132327b54131f2851f",
        "result_uncompressed_bytes": 26642,
        "result_compressed_sha256": null,
        "result_compressed_bytes": null,
        "result_gzip_contents_verified": false,
        "command": [
          "/usr/local/bin/python",
          "/workspace/cache/fixture.py",
          "--input",
          "/tmp/cache-run/input.npz",
          "--config",
          "/tmp/cache-run/config.json",
          "--output",
          "/tmp/cache-run/result.json",
          "--mode",
          "measure",
          "--scope",
          "task",
          "--repeats",
          "16"
        ],
        "capabilities": {
          "ncu_version": "NVIDIA (R) Nsight Compute Command Line Profiler\nCopyright (c) 2018-2025 NVIDIA Corporation\nVersion 2025.1.1.0 (build 35528883) (public-release)\n"
        },
        "returncode": 0,
        "workspace": ":16:8",
        "scope": "task",
        "status": "passed",
        "draw": 10,
        "predictions_bit_equal": true,
        "all_state_hashes_bit_equal": false,
        "state_fields_differing_from_original": [
          "final_parameter_sha256",
          "final_velocity_sha256",
          "state_vector_sha256",
          "scores_sha256"
        ],
        "state_fields_checked": [
          "input_hashes",
          "initial_parameter_sha256",
          "epoch_permutation_sha256",
          "final_parameter_sha256",
          "final_velocity_sha256",
          "state_vector_sha256",
          "predictions_sha256",
          "scores_sha256"
        ],
        "prediction_path": "qualify-10-result.predictions.npy",
        "prediction_sha256": "82cf23956ebfa10af72e8faa1ca0043bcc26492c17437708cebdc6b46a5a0717",
        "memory": {
          "peak_allocated_bytes": 15663104,
          "live_allocated_bytes": 14237696,
          "peak_reserved_bytes": 29360128,
          "named_storage_bytes": 13960864,
          "unnamed_live_allocated_bytes": 276832,
          "peak_allocated_mib": 14.9375,
          "nominal_l2_reference_bytes": 41943040,
          "peak_allocated_below_nominal_l2_capacity": true,
          "cache_residency_established": false,
          "scope": "Fresh task preparation including CUDA graph capture plus one complete reset/train/predict invocation; tensor allocator only, excluding driver/context. Recorded before output diagnostics and profiling warmup."
        },
        "software": {
          "python": "3.11.15",
          "torch": "2.5.1+cu124",
          "cuda": "12.4",
          "numpy": "2.2.6"
        }
      }
    ],
    "inherited_accuracy_valid": true,
    "draws_with_all_state_hashes_bit_equal": 0,
    "full_state_bit_equality_established": false,
    "equivalence_scope": "All class predictions, inputs, initial weights and epoch shuffles match. Different final-state or score hashes are explicitly retained; no numerical error bound is inferred from hashes.",
    "inherited_accuracy": {
      "correct": 98208,
      "total": 110000,
      "accuracy_percent": 89.28,
      "sample_sd_pp": 0.35908216329971143,
      "error_percent": 10.719999999999999,
      "meets_12_percent_error_target": true
    },
    "base_accuracy_sha256": "eb0abcd7613d987074b664ade1e4140699f298d5d58dfa5698b925e0071c7485"
  },
  "comparisons": {
    "baseline": {
      "result_path": "baseline.json",
      "result_sha256": "d3ad2a6e810459e53e9ecc3cacfdec761dca56ea1f6e8a97ad0d5dd66f056ef5",
      "result_uncompressed_sha256": "d3ad2a6e810459e53e9ecc3cacfdec761dca56ea1f6e8a97ad0d5dd66f056ef5",
      "result_uncompressed_bytes": 47833,
      "result_compressed_sha256": null,
      "result_compressed_bytes": null,
      "result_gzip_contents_verified": false,
      "command": [
        "/usr/local/bin/python",
        "/workspace/cache/fixture.py",
        "--input",
        "/tmp/cache-run/input.npz",
        "--config",
        "/tmp/cache-run/config.json",
        "--output",
        "/tmp/cache-run/result.json",
        "--mode",
        "measure",
        "--scope",
        "task",
        "--repeats",
        "16",
        "--energy"
      ],
      "capabilities": {
        "ncu_version": "NVIDIA (R) Nsight Compute Command Line Profiler\nCopyright (c) 2018-2025 NVIDIA Corporation\nVersion 2025.1.1.0 (build 35528883) (public-release)\n"
      },
      "returncode": 0,
      "workspace": ":4096:8",
      "scope": "task",
      "status": "passed",
      "draw": 0,
      "predictions_bit_equal": true,
      "all_state_hashes_bit_equal": true,
      "state_fields_differing_from_original": [],
      "state_fields_checked": [
        "input_hashes",
        "initial_parameter_sha256",
        "epoch_permutation_sha256",
        "final_parameter_sha256",
        "final_velocity_sha256",
        "state_vector_sha256",
        "predictions_sha256",
        "scores_sha256"
      ],
      "prediction_path": "baseline-result.predictions.npy",
      "prediction_sha256": "fce8a9458d2158537c7d6207fb0c8e9cfcbba007e7a748976bf3e7cd002729f3",
      "memory": {
        "peak_allocated_bytes": 82509824,
        "live_allocated_bytes": 81084416,
        "peak_reserved_bytes": 96468992,
        "named_storage_bytes": 13960864,
        "unnamed_live_allocated_bytes": 67123552,
        "peak_allocated_mib": 78.6875,
        "nominal_l2_reference_bytes": 41943040,
        "peak_allocated_below_nominal_l2_capacity": false,
        "cache_residency_established": false,
        "scope": "Fresh task preparation including CUDA graph capture plus one complete reset/train/predict invocation; tensor allocator only, excluding driver/context. Recorded before output diagnostics and profiling warmup."
      },
      "software": {
        "python": "3.11.15",
        "torch": "2.5.1+cu124",
        "cuda": "12.4",
        "numpy": "2.2.6"
      },
      "energy": {
        "passed": true,
        "trials": 3,
        "invocations_per_trial": 104,
        "raw_counter_arithmetic_recomputed": true,
        "summary": {
          "idle_adjusted_mj_per_task": {
            "mean": 1002.9797130200805,
            "sample_sd": 55.63196816552205,
            "values": [
              1059.7543262993806,
              1000.6192837807083,
              948.5655289801525
            ]
          },
          "unadjusted_mj_per_task": {
            "mean": 3305.477564102564,
            "sample_sd": 32.12786539688782,
            "values": [
              3299.423076923077,
              3340.201923076923,
              3276.8076923076924
            ]
          },
          "cuda_ms_per_task": {
            "mean": 37.90967422876602,
            "sample_sd": 0.1610706374635057,
            "values": [
              37.77359713040865,
              38.087512676532455,
              37.86791287935697
            ]
          },
          "wall_ms_per_task": {
            "mean": 37.94228616025641,
            "sample_sd": 0.1615132449200883,
            "values": [
              37.803932177884604,
              38.1197690144231,
              37.90315728846152
            ]
          }
        },
        "hardware": {
          "gpu": "NVIDIA A100-SXM4-40GB",
          "uuid": "GPU-059d2560-15ff-0871-a190-9065becce1b2",
          "total_memory_bytes": 42405855232,
          "power_limit_w": 400.0,
          "driver": "580.95.05",
          "cuda_pci_bus_id": "0000:15:00.0",
          "nvml_pci_bus_id": "00000000:15:00.0",
          "nvml_handle_selected_by_cuda_pci_bus_id": true
        },
        "scope": "Three paired-idle NVML trials for fresh complete tasks on draw zero"
      }
    },
    "small-workspace": {
      "result_path": "small-workspace.json",
      "result_sha256": "f935630bb4b48f2563e54834f065a1b7f442113fd2e645bd86c9ea1b7ded1e0b",
      "result_uncompressed_sha256": "f935630bb4b48f2563e54834f065a1b7f442113fd2e645bd86c9ea1b7ded1e0b",
      "result_uncompressed_bytes": 47828,
      "result_compressed_sha256": null,
      "result_compressed_bytes": null,
      "result_gzip_contents_verified": false,
      "command": [
        "/usr/local/bin/python",
        "/workspace/cache/fixture.py",
        "--input",
        "/tmp/cache-run/input.npz",
        "--config",
        "/tmp/cache-run/config.json",
        "--output",
        "/tmp/cache-run/result.json",
        "--mode",
        "measure",
        "--scope",
        "task",
        "--repeats",
        "16",
        "--energy"
      ],
      "capabilities": {
        "ncu_version": "NVIDIA (R) Nsight Compute Command Line Profiler\nCopyright (c) 2018-2025 NVIDIA Corporation\nVersion 2025.1.1.0 (build 35528883) (public-release)\n"
      },
      "returncode": 0,
      "workspace": ":16:8",
      "scope": "task",
      "status": "passed",
      "draw": 0,
      "predictions_bit_equal": true,
      "all_state_hashes_bit_equal": false,
      "state_fields_differing_from_original": [
        "final_parameter_sha256",
        "final_velocity_sha256",
        "state_vector_sha256",
        "scores_sha256"
      ],
      "state_fields_checked": [
        "input_hashes",
        "initial_parameter_sha256",
        "epoch_permutation_sha256",
        "final_parameter_sha256",
        "final_velocity_sha256",
        "state_vector_sha256",
        "predictions_sha256",
        "scores_sha256"
      ],
      "prediction_path": "small-workspace-result.predictions.npy",
      "prediction_sha256": "fce8a9458d2158537c7d6207fb0c8e9cfcbba007e7a748976bf3e7cd002729f3",
      "memory": {
        "peak_allocated_bytes": 15663104,
        "live_allocated_bytes": 14237696,
        "peak_reserved_bytes": 29360128,
        "named_storage_bytes": 13960864,
        "unnamed_live_allocated_bytes": 276832,
        "peak_allocated_mib": 14.9375,
        "nominal_l2_reference_bytes": 41943040,
        "peak_allocated_below_nominal_l2_capacity": true,
        "cache_residency_established": false,
        "scope": "Fresh task preparation including CUDA graph capture plus one complete reset/train/predict invocation; tensor allocator only, excluding driver/context. Recorded before output diagnostics and profiling warmup."
      },
      "software": {
        "python": "3.11.15",
        "torch": "2.5.1+cu124",
        "cuda": "12.4",
        "numpy": "2.2.6"
      },
      "energy": {
        "passed": true,
        "trials": 3,
        "invocations_per_trial": 108,
        "raw_counter_arithmetic_recomputed": true,
        "summary": {
          "idle_adjusted_mj_per_task": {
            "mean": 1018.4190392485816,
            "sample_sd": 47.31792745681788,
            "values": [
              1071.6258801087217,
              1002.5748159573074,
              981.0564216797158
            ]
          },
          "unadjusted_mj_per_task": {
            "mean": 3418.015432098766,
            "sample_sd": 30.297940864386373,
            "values": [
              3444.472222222222,
              3424.6111111111113,
              3384.962962962963
            ]
          },
          "cuda_ms_per_task": {
            "mean": 38.8594066478588,
            "sample_sd": 0.15432854365317664,
            "values": [
              38.87688078703704,
              39.00425437644676,
              38.697084780092595
            ]
          },
          "wall_ms_per_task": {
            "mean": 38.89153690740737,
            "sample_sd": 0.15269777054414221,
            "values": [
              38.90498400462961,
              39.037066407407345,
              38.73256031018515
            ]
          }
        },
        "hardware": {
          "gpu": "NVIDIA A100-SXM4-40GB",
          "uuid": "GPU-87335ec7-4260-ce3c-9bea-69a4ecbf5771",
          "total_memory_bytes": 42405855232,
          "power_limit_w": 400.0,
          "driver": "580.95.05",
          "cuda_pci_bus_id": "0000:0F:00.0",
          "nvml_pci_bus_id": "00000000:0F:00.0",
          "nvml_handle_selected_by_cuda_pci_bus_id": true
        },
        "scope": "Three paired-idle NVML trials for fresh complete tasks on draw zero"
      }
    },
    "zero-workspace": {
      "result_path": "zero-workspace.json",
      "result_sha256": "f2bfb50e4594c151bca7b3014419f9b3d35180b76e02b01096fa57f1d4a3d79d",
      "result_uncompressed_sha256": "f2bfb50e4594c151bca7b3014419f9b3d35180b76e02b01096fa57f1d4a3d79d",
      "result_uncompressed_bytes": 47798,
      "result_compressed_sha256": null,
      "result_compressed_bytes": null,
      "result_gzip_contents_verified": false,
      "command": [
        "/usr/local/bin/python",
        "/workspace/cache/fixture.py",
        "--input",
        "/tmp/cache-run/input.npz",
        "--config",
        "/tmp/cache-run/config.json",
        "--output",
        "/tmp/cache-run/result.json",
        "--mode",
        "measure",
        "--scope",
        "task",
        "--repeats",
        "16",
        "--energy"
      ],
      "capabilities": {
        "ncu_version": "NVIDIA (R) Nsight Compute Command Line Profiler\nCopyright (c) 2018-2025 NVIDIA Corporation\nVersion 2025.1.1.0 (build 35528883) (public-release)\n"
      },
      "returncode": 0,
      "workspace": ":0:0",
      "scope": "task",
      "status": "passed",
      "draw": 0,
      "predictions_bit_equal": true,
      "all_state_hashes_bit_equal": false,
      "state_fields_differing_from_original": [
        "final_parameter_sha256",
        "final_velocity_sha256",
        "state_vector_sha256",
        "scores_sha256"
      ],
      "state_fields_checked": [
        "input_hashes",
        "initial_parameter_sha256",
        "epoch_permutation_sha256",
        "final_parameter_sha256",
        "final_velocity_sha256",
        "state_vector_sha256",
        "predictions_sha256",
        "scores_sha256"
      ],
      "prediction_path": "zero-workspace-result.predictions.npy",
      "prediction_sha256": "fce8a9458d2158537c7d6207fb0c8e9cfcbba007e7a748976bf3e7cd002729f3",
      "memory": {
        "peak_allocated_bytes": 15400960,
        "live_allocated_bytes": 13975552,
        "peak_reserved_bytes": 29360128,
        "named_storage_bytes": 13960864,
        "unnamed_live_allocated_bytes": 14688,
        "peak_allocated_mib": 14.6875,
        "nominal_l2_reference_bytes": 41943040,
        "peak_allocated_below_nominal_l2_capacity": true,
        "cache_residency_established": false,
        "scope": "Fresh task preparation including CUDA graph capture plus one complete reset/train/predict invocation; tensor allocator only, excluding driver/context. Recorded before output diagnostics and profiling warmup."
      },
      "software": {
        "python": "3.11.15",
        "torch": "2.5.1+cu124",
        "cuda": "12.4",
        "numpy": "2.2.6"
      },
      "energy": {
        "passed": true,
        "trials": 3,
        "invocations_per_trial": 124,
        "raw_counter_arithmetic_recomputed": true,
        "summary": {
          "idle_adjusted_mj_per_task": {
            "mean": 994.918786255717,
            "sample_sd": 5.529763585844365,
            "values": [
              989.2644849464956,
              995.1769015186421,
              1000.3149723020133
            ]
          },
          "unadjusted_mj_per_task": {
            "mean": 3203.755376344086,
            "sample_sd": 17.07544761864355,
            "values": [
              3186.282258064516,
              3204.5806451612902,
              3220.4032258064517
            ]
          },
          "cuda_ms_per_task": {
            "mean": 37.375288768481184,
            "sample_sd": 0.04146411220358215,
            "values": [
              37.33301568800403,
              37.37695706275202,
              37.4158935546875
            ]
          },
          "wall_ms_per_task": {
            "mean": 37.40340780913981,
            "sample_sd": 0.04254968975731325,
            "values": [
              37.35751584274191,
              37.41115768548389,
              37.44154989919362
            ]
          }
        },
        "hardware": {
          "gpu": "NVIDIA A100-SXM4-40GB",
          "uuid": "GPU-395167ee-ee3d-1e5b-4df9-d552d25cd56f",
          "total_memory_bytes": 42405855232,
          "power_limit_w": 400.0,
          "driver": "580.95.05",
          "cuda_pci_bus_id": "0000:DA:00.0",
          "nvml_pci_bus_id": "00000000:DA:00.0",
          "nvml_handle_selected_by_cuda_pci_bus_id": true
        },
        "scope": "Three paired-idle NVML trials for fresh complete tasks on draw zero"
      }
    }
  },
  "reductions": {
    "small-workspace": {
      "peak_allocated_bytes_saved": 66846720,
      "peak_allocated_reduction_percent": 81.01667990468626,
      "peak_allocated_ratio_to_baseline": 0.1898332009531374,
      "idle_adjusted_energy_ratio_to_baseline": 1.0153934581408548
    },
    "zero-workspace": {
      "peak_allocated_bytes_saved": 67108864,
      "peak_allocated_reduction_percent": 81.33439237490072,
      "peak_allocated_ratio_to_baseline": 0.18665607625099284,
      "idle_adjusted_energy_ratio_to_baseline": 0.9919630211262288
    }
  },
  "profiles": {
    "profile-baseline-task": {
      "result_path": "profile-baseline-task.json.gz",
      "result_sha256": "412620275c5cb7c8c5bc7617639d3bda0ac4fb9eed40a80cdede2e9655a5a00c",
      "result_uncompressed_sha256": "03d14d11ef00cd7aee383fe06ea5c639276bea5e4cb29e49f813214b6430648a",
      "result_uncompressed_bytes": 80370913,
      "result_compressed_sha256": "412620275c5cb7c8c5bc7617639d3bda0ac4fb9eed40a80cdede2e9655a5a00c",
      "result_compressed_bytes": 1587219,
      "result_gzip_contents_verified": true,
      "command": [
        "/opt/nvidia/nsight-compute/2025.1.1/ncu",
        "--replay-mode",
        "app-range",
        "--cache-control",
        "none",
        "--clock-control",
        "none",
        "--metrics",
        "dram__bytes_read.sum,dram__bytes_write.sum,lts__t_sectors_op_read.sum,lts__t_sectors_op_write.sum,lts__t_sector_hit_rate.pct",
        "--csv",
        "--page",
        "raw",
        "--force-overwrite",
        "--export",
        "/tmp/cache-run/profile",
        "/usr/local/bin/python",
        "/workspace/cache/fixture.py",
        "--input",
        "/tmp/cache-run/input.npz",
        "--config",
        "/tmp/cache-run/config.json",
        "--output",
        "/tmp/cache-run/result.json",
        "--mode",
        "profile",
        "--scope",
        "task",
        "--repeats",
        "16"
      ],
      "capabilities": {
        "ncu_version": "NVIDIA (R) Nsight Compute Command Line Profiler\nCopyright (c) 2018-2025 NVIDIA Corporation\nVersion 2025.1.1.0 (build 35528883) (public-release)\n",
        "metric_query": {
          "returncode": 0,
          "raw_query_sha256": "d985c130f3192a845ada05044ebdcac1df547c7bbaa3b3f5a5c712414433125e",
          "raw_query_retained_in": "profile-baseline-task.json.gz"
        }
      },
      "returncode": 0,
      "workspace": ":4096:8",
      "scope": "task",
      "status": "passed",
      "draw": 0,
      "predictions_bit_equal": true,
      "all_state_hashes_bit_equal": true,
      "state_fields_differing_from_original": [],
      "state_fields_checked": [
        "input_hashes",
        "initial_parameter_sha256",
        "epoch_permutation_sha256",
        "final_parameter_sha256",
        "final_velocity_sha256",
        "state_vector_sha256",
        "predictions_sha256",
        "scores_sha256"
      ],
      "prediction_path": "profile-baseline-task-result.predictions.npy",
      "prediction_sha256": "fce8a9458d2158537c7d6207fb0c8e9cfcbba007e7a748976bf3e7cd002729f3",
      "memory": {
        "peak_allocated_bytes": 82509824,
        "live_allocated_bytes": 81084416,
        "peak_reserved_bytes": 96468992,
        "named_storage_bytes": 13960864,
        "unnamed_live_allocated_bytes": 67123552,
        "peak_allocated_mib": 78.6875,
        "nominal_l2_reference_bytes": 41943040,
        "peak_allocated_below_nominal_l2_capacity": false,
        "cache_residency_established": false,
        "scope": "Fresh task preparation including CUDA graph capture plus one complete reset/train/predict invocation; tensor allocator only, excluding driver/context. Recorded before output diagnostics and profiling warmup."
      },
      "software": {
        "python": "3.11.15",
        "torch": "2.5.1+cu124",
        "cuda": "12.4",
        "numpy": "2.2.6"
      },
      "dram_traffic_measured": true,
      "profile": {
        "scope": "task",
        "warmup_complete_tasks": 2,
        "requested_repeats": 16,
        "fresh_tasks_in_range": 16,
        "fresh_training_sequences_in_range": 16,
        "epochs_per_sequence": 2,
        "training_scope_ignores_repeats": false,
        "reset_inside_range": true,
        "prediction_inside_range": true,
        "epoch_order_copies_inside_range": true,
        "output_serialization_inside_range": false,
        "cpu_validation_inside_range": false,
        "synchronized_before_stop": true
      },
      "ncu_report_path": "profile-baseline-task-profile.ncu-rep.gz",
      "ncu_report_sha256": "d4d72f72149a32ac9285e48294e7c2b8b42b5adfb3ff3fd82e1a9ec495e0aeb9",
      "ncu_report_uncompressed_sha256": "d63ce9073e7b4564cf1de5b514d1823bdb573fa9cfede2dd1051bc0a3fcd2f66",
      "ncu_report_uncompressed_bytes": 88576540,
      "ncu_report_compressed_sha256": "d4d72f72149a32ac9285e48294e7c2b8b42b5adfb3ff3fd82e1a9ec495e0aeb9",
      "ncu_report_compressed_bytes": 11959402,
      "ncu_report_gzip_contents_verified": true,
      "metric_rows": {
        "dram__bytes_read.sum": {
          "value": 49300864.0,
          "raw_value": "49300864",
          "raw_unit": "byte",
          "range_id": "0"
        },
        "dram__bytes_write.sum": {
          "value": 29329408.0,
          "raw_value": "29329408",
          "raw_unit": "byte",
          "range_id": "0"
        },
        "lts__t_sectors_op_read.sum": {
          "value": 474067036.0,
          "raw_value": "474067036",
          "raw_unit": "sector",
          "range_id": "0"
        },
        "lts__t_sectors_op_write.sum": {
          "value": 237557538.0,
          "raw_value": "237557538",
          "raw_unit": "sector",
          "range_id": "0"
        },
        "lts__t_sector_hit_rate.pct": {
          "value": 95.49,
          "raw_value": "95.49",
          "raw_unit": "%",
          "range_id": "0"
        }
      },
      "pass_counts": [
        2
      ],
      "profile_device_attributes": {
        "device__attribute_display_name": "NVIDIA A100-SXM4-80GB",
        "device__attribute_l2_cache_size": "41943040",
        "device__attribute_total_memory": "85094825984",
        "device__attribute_memory_clock_rate": "1593000",
        "device__attribute_multiprocessor_count": "108",
        "profiler__replayer_passes": "2",
        "profiler__replayer_passes_type_warmup": "0",
        "gpu__time_duration.sum": "2584336832"
      },
      "pass_count_retained": true,
      "range_training_sequences": 16,
      "dram_read_bytes_per_sequence": 3081304.0,
      "dram_write_bytes_per_sequence": 1833088.0,
      "dram_total_bytes_per_sequence": 4914392.0,
      "l2_read_request_bytes_per_sequence": 948134072.0,
      "l2_write_request_bytes_per_sequence": 475115076.0,
      "l2_sector_hit_rate_percent": 95.49,
      "counter_unit": "per complete fresh reset/train/predict task"
    },
    "profile-baseline-training": {
      "result_path": "profile-baseline-training.json.gz",
      "result_sha256": "426de66111db1a5fd1e9b4b5d4ea7e2ce56ceb3570aa23c882977edfaa1e8431",
      "result_uncompressed_sha256": "5aebbd9227c77209ff8cbc6693ed96be7d4e574b2fed9aff15a88ab58f37c2df",
      "result_uncompressed_bytes": 80370909,
      "result_compressed_sha256": "426de66111db1a5fd1e9b4b5d4ea7e2ce56ceb3570aa23c882977edfaa1e8431",
      "result_compressed_bytes": 1587229,
      "result_gzip_contents_verified": true,
      "command": [
        "/opt/nvidia/nsight-compute/2025.1.1/ncu",
        "--replay-mode",
        "app-range",
        "--cache-control",
        "none",
        "--clock-control",
        "none",
        "--metrics",
        "dram__bytes_read.sum,dram__bytes_write.sum,lts__t_sectors_op_read.sum,lts__t_sectors_op_write.sum,lts__t_sector_hit_rate.pct",
        "--csv",
        "--page",
        "raw",
        "--force-overwrite",
        "--export",
        "/tmp/cache-run/profile",
        "/usr/local/bin/python",
        "/workspace/cache/fixture.py",
        "--input",
        "/tmp/cache-run/input.npz",
        "--config",
        "/tmp/cache-run/config.json",
        "--output",
        "/tmp/cache-run/result.json",
        "--mode",
        "profile",
        "--scope",
        "training",
        "--repeats",
        "16"
      ],
      "capabilities": {
        "ncu_version": "NVIDIA (R) Nsight Compute Command Line Profiler\nCopyright (c) 2018-2025 NVIDIA Corporation\nVersion 2025.1.1.0 (build 35528883) (public-release)\n",
        "metric_query": {
          "returncode": 0,
          "raw_query_sha256": "adaab0cf189570bd1f0e332014f1dd49b40842687eedc0929b6867e8723c54c5",
          "raw_query_retained_in": "profile-baseline-training.json.gz"
        }
      },
      "returncode": 0,
      "workspace": ":4096:8",
      "scope": "training",
      "status": "passed",
      "draw": 0,
      "predictions_bit_equal": true,
      "all_state_hashes_bit_equal": true,
      "state_fields_differing_from_original": [],
      "state_fields_checked": [
        "input_hashes",
        "initial_parameter_sha256",
        "epoch_permutation_sha256",
        "final_parameter_sha256",
        "final_velocity_sha256",
        "state_vector_sha256",
        "predictions_sha256",
        "scores_sha256"
      ],
      "prediction_path": "profile-baseline-training-result.predictions.npy",
      "prediction_sha256": "fce8a9458d2158537c7d6207fb0c8e9cfcbba007e7a748976bf3e7cd002729f3",
      "memory": {
        "peak_allocated_bytes": 82509824,
        "live_allocated_bytes": 81084416,
        "peak_reserved_bytes": 96468992,
        "named_storage_bytes": 13960864,
        "unnamed_live_allocated_bytes": 67123552,
        "peak_allocated_mib": 78.6875,
        "nominal_l2_reference_bytes": 41943040,
        "peak_allocated_below_nominal_l2_capacity": false,
        "cache_residency_established": false,
        "scope": "Fresh task preparation including CUDA graph capture plus one complete reset/train/predict invocation; tensor allocator only, excluding driver/context. Recorded before output diagnostics and profiling warmup."
      },
      "software": {
        "python": "3.11.15",
        "torch": "2.5.1+cu124",
        "cuda": "12.4",
        "numpy": "2.2.6"
      },
      "dram_traffic_measured": true,
      "profile": {
        "scope": "training",
        "warmup_complete_tasks": 2,
        "requested_repeats": 16,
        "fresh_tasks_in_range": 0,
        "fresh_training_sequences_in_range": 1,
        "epochs_per_sequence": 2,
        "training_scope_ignores_repeats": true,
        "reset_inside_range": false,
        "prediction_inside_range": false,
        "epoch_order_copies_inside_range": true,
        "output_serialization_inside_range": false,
        "cpu_validation_inside_range": false,
        "synchronized_before_stop": true
      },
      "ncu_report_path": "profile-baseline-training-profile.ncu-rep.gz",
      "ncu_report_sha256": "8942e31975b4da1c2930d3f8c2e6e60f51bccf7da1a5ae614356355b7da9679b",
      "ncu_report_uncompressed_sha256": "edf26ff2704e497147a22047d01225e3953b7b4a913e49c63a4f3c633ef6976c",
      "ncu_report_uncompressed_bytes": 6742899,
      "ncu_report_compressed_sha256": "8942e31975b4da1c2930d3f8c2e6e60f51bccf7da1a5ae614356355b7da9679b",
      "ncu_report_compressed_bytes": 870848,
      "ncu_report_gzip_contents_verified": true,
      "metric_rows": {
        "dram__bytes_read.sum": {
          "value": 2265984.0,
          "raw_value": "2265984",
          "raw_unit": "byte",
          "range_id": "0"
        },
        "dram__bytes_write.sum": {
          "value": 1896832.0,
          "raw_value": "1896832",
          "raw_unit": "byte",
          "range_id": "0"
        },
        "lts__t_sectors_op_read.sum": {
          "value": 26466194.0,
          "raw_value": "26466194",
          "raw_unit": "sector",
          "range_id": "0"
        },
        "lts__t_sectors_op_write.sum": {
          "value": 12477848.0,
          "raw_value": "12477848",
          "raw_unit": "sector",
          "range_id": "0"
        },
        "lts__t_sector_hit_rate.pct": {
          "value": 95.29,
          "raw_value": "95.29",
          "raw_unit": "%",
          "range_id": "0"
        }
      },
      "pass_counts": [
        2
      ],
      "profile_device_attributes": {
        "device__attribute_display_name": "NVIDIA A100-SXM4-40GB",
        "device__attribute_l2_cache_size": "41943040",
        "device__attribute_total_memory": "42405855232",
        "device__attribute_memory_clock_rate": "1215000",
        "device__attribute_multiprocessor_count": "108",
        "profiler__replayer_passes": "2",
        "profiler__replayer_passes_type_warmup": "0",
        "gpu__time_duration.sum": "144168640"
      },
      "pass_count_retained": true,
      "range_training_sequences": 1,
      "dram_read_bytes_per_sequence": 2265984.0,
      "dram_write_bytes_per_sequence": 1896832.0,
      "dram_total_bytes_per_sequence": 4162816.0,
      "l2_read_request_bytes_per_sequence": 846918208.0,
      "l2_write_request_bytes_per_sequence": 399291136.0,
      "l2_sector_hit_rate_percent": 95.29,
      "counter_unit": "per original two-epoch training sequence, reset and prediction outside range"
    },
    "profile-small-task": {
      "result_path": "profile-small-task.json.gz",
      "result_sha256": "e7c527c7b1ada773d6b176f30e84c97e36886f3521558de8f80ab2040c56c23c",
      "result_uncompressed_sha256": "1a4cd304ffb3d28ee33b841da45b404bd0d47c00c35366cb134f21439c6d0c51",
      "result_uncompressed_bytes": 80370903,
      "result_compressed_sha256": "e7c527c7b1ada773d6b176f30e84c97e36886f3521558de8f80ab2040c56c23c",
      "result_compressed_bytes": 1587223,
      "result_gzip_contents_verified": true,
      "command": [
        "/opt/nvidia/nsight-compute/2025.1.1/ncu",
        "--replay-mode",
        "app-range",
        "--cache-control",
        "none",
        "--clock-control",
        "none",
        "--metrics",
        "dram__bytes_read.sum,dram__bytes_write.sum,lts__t_sectors_op_read.sum,lts__t_sectors_op_write.sum,lts__t_sector_hit_rate.pct",
        "--csv",
        "--page",
        "raw",
        "--force-overwrite",
        "--export",
        "/tmp/cache-run/profile",
        "/usr/local/bin/python",
        "/workspace/cache/fixture.py",
        "--input",
        "/tmp/cache-run/input.npz",
        "--config",
        "/tmp/cache-run/config.json",
        "--output",
        "/tmp/cache-run/result.json",
        "--mode",
        "profile",
        "--scope",
        "task",
        "--repeats",
        "16"
      ],
      "capabilities": {
        "ncu_version": "NVIDIA (R) Nsight Compute Command Line Profiler\nCopyright (c) 2018-2025 NVIDIA Corporation\nVersion 2025.1.1.0 (build 35528883) (public-release)\n",
        "metric_query": {
          "returncode": 0,
          "raw_query_sha256": "adaab0cf189570bd1f0e332014f1dd49b40842687eedc0929b6867e8723c54c5",
          "raw_query_retained_in": "profile-small-task.json.gz"
        }
      },
      "returncode": 0,
      "workspace": ":16:8",
      "scope": "task",
      "status": "passed",
      "draw": 0,
      "predictions_bit_equal": true,
      "all_state_hashes_bit_equal": false,
      "state_fields_differing_from_original": [
        "final_parameter_sha256",
        "final_velocity_sha256",
        "state_vector_sha256",
        "scores_sha256"
      ],
      "state_fields_checked": [
        "input_hashes",
        "initial_parameter_sha256",
        "epoch_permutation_sha256",
        "final_parameter_sha256",
        "final_velocity_sha256",
        "state_vector_sha256",
        "predictions_sha256",
        "scores_sha256"
      ],
      "prediction_path": "profile-small-task-result.predictions.npy",
      "prediction_sha256": "fce8a9458d2158537c7d6207fb0c8e9cfcbba007e7a748976bf3e7cd002729f3",
      "memory": {
        "peak_allocated_bytes": 15663104,
        "live_allocated_bytes": 14237696,
        "peak_reserved_bytes": 29360128,
        "named_storage_bytes": 13960864,
        "unnamed_live_allocated_bytes": 276832,
        "peak_allocated_mib": 14.9375,
        "nominal_l2_reference_bytes": 41943040,
        "peak_allocated_below_nominal_l2_capacity": true,
        "cache_residency_established": false,
        "scope": "Fresh task preparation including CUDA graph capture plus one complete reset/train/predict invocation; tensor allocator only, excluding driver/context. Recorded before output diagnostics and profiling warmup."
      },
      "software": {
        "python": "3.11.15",
        "torch": "2.5.1+cu124",
        "cuda": "12.4",
        "numpy": "2.2.6"
      },
      "dram_traffic_measured": true,
      "profile": {
        "scope": "task",
        "warmup_complete_tasks": 2,
        "requested_repeats": 16,
        "fresh_tasks_in_range": 16,
        "fresh_training_sequences_in_range": 16,
        "epochs_per_sequence": 2,
        "training_scope_ignores_repeats": false,
        "reset_inside_range": true,
        "prediction_inside_range": true,
        "epoch_order_copies_inside_range": true,
        "output_serialization_inside_range": false,
        "cpu_validation_inside_range": false,
        "synchronized_before_stop": true
      },
      "ncu_report_path": "profile-small-task-profile.ncu-rep.gz",
      "ncu_report_sha256": "b23501d03bba541bf48a02cbbf09481762144553f69b09654e99f93d33aee4f8",
      "ncu_report_uncompressed_sha256": "bd4bb547c58a0625203d5992fe5bf0830cb322022f627ba272899660cf849afa",
      "ncu_report_uncompressed_bytes": 85003041,
      "ncu_report_compressed_sha256": "b23501d03bba541bf48a02cbbf09481762144553f69b09654e99f93d33aee4f8",
      "ncu_report_compressed_bytes": 11416744,
      "ncu_report_gzip_contents_verified": true,
      "metric_rows": {
        "dram__bytes_read.sum": {
          "value": 52088576.0,
          "raw_value": "52088576",
          "raw_unit": "byte",
          "range_id": "0"
        },
        "dram__bytes_write.sum": {
          "value": 29900800.0,
          "raw_value": "29900800",
          "raw_unit": "byte",
          "range_id": "0"
        },
        "lts__t_sectors_op_read.sum": {
          "value": 429809710.0,
          "raw_value": "429809710",
          "raw_unit": "sector",
          "range_id": "0"
        },
        "lts__t_sectors_op_write.sum": {
          "value": 226110828.0,
          "raw_value": "226110828",
          "raw_unit": "sector",
          "range_id": "0"
        },
        "lts__t_sector_hit_rate.pct": {
          "value": 95.45,
          "raw_value": "95.45",
          "raw_unit": "%",
          "range_id": "0"
        }
      },
      "pass_counts": [
        2
      ],
      "profile_device_attributes": {
        "device__attribute_display_name": "NVIDIA A100-SXM4-40GB",
        "device__attribute_l2_cache_size": "41943040",
        "device__attribute_total_memory": "42405855232",
        "device__attribute_memory_clock_rate": "1215000",
        "device__attribute_multiprocessor_count": "108",
        "profiler__replayer_passes": "2",
        "profiler__replayer_passes_type_warmup": "0",
        "gpu__time_duration.sum": "2451604224"
      },
      "pass_count_retained": true,
      "range_training_sequences": 16,
      "dram_read_bytes_per_sequence": 3255536.0,
      "dram_write_bytes_per_sequence": 1868800.0,
      "dram_total_bytes_per_sequence": 5124336.0,
      "l2_read_request_bytes_per_sequence": 859619420.0,
      "l2_write_request_bytes_per_sequence": 452221656.0,
      "l2_sector_hit_rate_percent": 95.45,
      "counter_unit": "per complete fresh reset/train/predict task"
    },
    "profile-small-training": {
      "result_path": "profile-small-training.json.gz",
      "result_sha256": "ebb2dfa1fd5162034234f0fb064b6a51ee58965a0682cab88776a7f8a0b47f5c",
      "result_uncompressed_sha256": "c2a2d33a2a0f4a823a7a24fb5643d0ae23d01ead2920dd1832ec0dbc66bace8e",
      "result_uncompressed_bytes": 80370901,
      "result_compressed_sha256": "ebb2dfa1fd5162034234f0fb064b6a51ee58965a0682cab88776a7f8a0b47f5c",
      "result_compressed_bytes": 1587197,
      "result_gzip_contents_verified": true,
      "command": [
        "/opt/nvidia/nsight-compute/2025.1.1/ncu",
        "--replay-mode",
        "app-range",
        "--cache-control",
        "none",
        "--clock-control",
        "none",
        "--metrics",
        "dram__bytes_read.sum,dram__bytes_write.sum,lts__t_sectors_op_read.sum,lts__t_sectors_op_write.sum,lts__t_sector_hit_rate.pct",
        "--csv",
        "--page",
        "raw",
        "--force-overwrite",
        "--export",
        "/tmp/cache-run/profile",
        "/usr/local/bin/python",
        "/workspace/cache/fixture.py",
        "--input",
        "/tmp/cache-run/input.npz",
        "--config",
        "/tmp/cache-run/config.json",
        "--output",
        "/tmp/cache-run/result.json",
        "--mode",
        "profile",
        "--scope",
        "training",
        "--repeats",
        "16"
      ],
      "capabilities": {
        "ncu_version": "NVIDIA (R) Nsight Compute Command Line Profiler\nCopyright (c) 2018-2025 NVIDIA Corporation\nVersion 2025.1.1.0 (build 35528883) (public-release)\n",
        "metric_query": {
          "returncode": 0,
          "raw_query_sha256": "adaab0cf189570bd1f0e332014f1dd49b40842687eedc0929b6867e8723c54c5",
          "raw_query_retained_in": "profile-small-training.json.gz"
        }
      },
      "returncode": 0,
      "workspace": ":16:8",
      "scope": "training",
      "status": "passed",
      "draw": 0,
      "predictions_bit_equal": true,
      "all_state_hashes_bit_equal": false,
      "state_fields_differing_from_original": [
        "final_parameter_sha256",
        "final_velocity_sha256",
        "state_vector_sha256",
        "scores_sha256"
      ],
      "state_fields_checked": [
        "input_hashes",
        "initial_parameter_sha256",
        "epoch_permutation_sha256",
        "final_parameter_sha256",
        "final_velocity_sha256",
        "state_vector_sha256",
        "predictions_sha256",
        "scores_sha256"
      ],
      "prediction_path": "profile-small-training-result.predictions.npy",
      "prediction_sha256": "fce8a9458d2158537c7d6207fb0c8e9cfcbba007e7a748976bf3e7cd002729f3",
      "memory": {
        "peak_allocated_bytes": 15663104,
        "live_allocated_bytes": 14237696,
        "peak_reserved_bytes": 29360128,
        "named_storage_bytes": 13960864,
        "unnamed_live_allocated_bytes": 276832,
        "peak_allocated_mib": 14.9375,
        "nominal_l2_reference_bytes": 41943040,
        "peak_allocated_below_nominal_l2_capacity": true,
        "cache_residency_established": false,
        "scope": "Fresh task preparation including CUDA graph capture plus one complete reset/train/predict invocation; tensor allocator only, excluding driver/context. Recorded before output diagnostics and profiling warmup."
      },
      "software": {
        "python": "3.11.15",
        "torch": "2.5.1+cu124",
        "cuda": "12.4",
        "numpy": "2.2.6"
      },
      "dram_traffic_measured": true,
      "profile": {
        "scope": "training",
        "warmup_complete_tasks": 2,
        "requested_repeats": 16,
        "fresh_tasks_in_range": 0,
        "fresh_training_sequences_in_range": 1,
        "epochs_per_sequence": 2,
        "training_scope_ignores_repeats": true,
        "reset_inside_range": false,
        "prediction_inside_range": false,
        "epoch_order_copies_inside_range": true,
        "output_serialization_inside_range": false,
        "cpu_validation_inside_range": false,
        "synchronized_before_stop": true
      },
      "ncu_report_path": "profile-small-training-profile.ncu-rep.gz",
      "ncu_report_sha256": "e7171b7daf0cfa33b3dc594db3539f64b0bc198f57cdec570e9b5ac9dc7257fc",
      "ncu_report_uncompressed_sha256": "eb058c1179676d6c709258997877650be716f1f682a0acd1d3c3895b37c38b3c",
      "ncu_report_uncompressed_bytes": 6534891,
      "ncu_report_compressed_sha256": "e7171b7daf0cfa33b3dc594db3539f64b0bc198f57cdec570e9b5ac9dc7257fc",
      "ncu_report_compressed_bytes": 836089,
      "ncu_report_gzip_contents_verified": true,
      "metric_rows": {
        "dram__bytes_read.sum": {
          "value": 2036480.0,
          "raw_value": "2036480",
          "raw_unit": "byte",
          "range_id": "0"
        },
        "dram__bytes_write.sum": {
          "value": 1595648.0,
          "raw_value": "1595648",
          "raw_unit": "byte",
          "range_id": "0"
        },
        "lts__t_sectors_op_read.sum": {
          "value": 23587372.0,
          "raw_value": "23587372",
          "raw_unit": "sector",
          "range_id": "0"
        },
        "lts__t_sectors_op_write.sum": {
          "value": 11740492.0,
          "raw_value": "11740492",
          "raw_unit": "sector",
          "range_id": "0"
        },
        "lts__t_sector_hit_rate.pct": {
          "value": 97.58,
          "raw_value": "97.58",
          "raw_unit": "%",
          "range_id": "0"
        }
      },
      "pass_counts": [
        2
      ],
      "profile_device_attributes": {
        "device__attribute_display_name": "NVIDIA A100-SXM4-40GB",
        "device__attribute_l2_cache_size": "41943040",
        "device__attribute_total_memory": "42405855232",
        "device__attribute_memory_clock_rate": "1215000",
        "device__attribute_multiprocessor_count": "108",
        "profiler__replayer_passes": "2",
        "profiler__replayer_passes_type_warmup": "0",
        "gpu__time_duration.sum": "149245664"
      },
      "pass_count_retained": true,
      "range_training_sequences": 1,
      "dram_read_bytes_per_sequence": 2036480.0,
      "dram_write_bytes_per_sequence": 1595648.0,
      "dram_total_bytes_per_sequence": 3632128.0,
      "l2_read_request_bytes_per_sequence": 754795904.0,
      "l2_write_request_bytes_per_sequence": 375695744.0,
      "l2_sector_hit_rate_percent": 97.58,
      "counter_unit": "per original two-epoch training sequence, reset and prediction outside range"
    }
  },
  "scope": "Independent CPU comparison against frozen predictions and state, storage accounting, and retained NVML/Ncu counters. No test labels or GPU retraining used.",
  "cache_claim": "Capacity and measured traffic are separate. No claim that every access hits L2 or that profiling proves universal cache residency."
}
