{
  "report_date": "2026-09-02",
  "joules_per_operand_grid_step": 1e-15,
  "movement_energy_unit": "joules, using 1e-15 J per source operand per abstract address-grid step",
  "distance": "d(address) = ceil(sqrt(address))",
  "int8_step_interpretation": {
    "dally_2023_on_chip_communication_fJ_per_bit_mm": 100,
    "abstract_step_energy_J_per_int8_operand": 1e-15,
    "equivalent_distance_um_per_step": 1.25,
    "qualification": "an energy-equivalent scale comparison, not a claim that an abstract address step is a physical 1.25 um wire"
  },
  "network": {
    "dimensions": [
      784,
      2500,
      2000,
      1500,
      1000,
      500,
      10
    ],
    "examples_per_inference_epoch": 60000,
    "logical_MAC_per_example": 11965000,
    "logical_MAC_per_epoch": 717900000000
  },
  "schedule": {
    "family": "jointly packed generalized rectangular sa_cache",
    "tile_search": "best of three deterministic coordinate-descent starts: independent layer optima, all-minimum tiles, and all-maximum tiles",
    "qualification": "best found within this schedule family; not a global lower bound"
  },
  "batching_analysis": {
    "persistent_epoch_address_space": {
      "description": "weights, the 60k-image source, and the 60k-image output are allocated once and retained; layer-specific batch-local scratch and intermediate activation buffers are reused across invocations",
      "input_storage": {
        "materialized_before_modeled_run": true,
        "examples": 60000,
        "features_per_example": 784,
        "distinct_input_cells": 47040000,
        "reusable_64_row_input_buffer": false,
        "batch64_partition": [
          {
            "calls": 937,
            "rows_per_call": 64,
            "distinct_input_cells": 47014912
          },
          {
            "calls": 1,
            "rows_per_call": 32,
            "distinct_input_cells": 25088
          }
        ],
        "address_assignment": "all input cells are jointly frequency-packed in the persistent epoch address space",
        "initial_materialization_energy": "excluded; inputs exist at their assigned addresses before modeled execution begins"
      },
      "transient_storage": "computed hidden-layer activations use layer-specific batch-row buffers that are overwritten across invocations",
      "cases": {
        "full": {
          "batch_size": 60000,
          "batch_invocations": 1,
          "decomposition": [
            {
              "rows": 60000,
              "invocation_count": 1
            }
          ],
          "epoch_charged_reads": 2888892000000,
          "movement_intensity_grid_steps_per_charged_read": 46.21247718972291,
          "epoch_movement_energy_J": 0.133502855653573,
          "dally_2023_all_charged_reads_small_RAM_same_schedule_sensitivity_J": 1.2890596556535732,
          "coordinate_descent_sweeps": 4,
          "coordinate_descent_starts": 3,
          "winning_start": "all_minimum_tiles",
          "tiles_by_invocation_rows": {
            "60000": [
              {
                "layer": 0,
                "input_width": 784,
                "output_width": 2500,
                "tile_i": 96,
                "tile_j": 125
              },
              {
                "layer": 1,
                "input_width": 2500,
                "output_width": 2000,
                "tile_i": 60,
                "tile_j": 250
              },
              {
                "layer": 2,
                "input_width": 2000,
                "output_width": 1500,
                "tile_i": 48,
                "tile_j": 375
              },
              {
                "layer": 3,
                "input_width": 1500,
                "output_width": 1000,
                "tile_i": 30,
                "tile_j": 500
              },
              {
                "layer": 4,
                "input_width": 1000,
                "output_width": 500,
                "tile_i": 16,
                "tile_j": 500
              },
              {
                "layer": 5,
                "input_width": 500,
                "output_width": 10,
                "tile_i": 250,
                "tile_j": 10
              }
            ]
          },
          "movement_energy_ratio_to_full_batch": 1.0
        },
        "64": {
          "batch_size": 64,
          "batch_invocations": 938,
          "decomposition": [
            {
              "rows": 64,
              "invocation_count": 937
            },
            {
              "rows": 32,
              "invocation_count": 1
            }
          ],
          "epoch_charged_reads": 2909028890000,
          "movement_intensity_grid_steps_per_charged_read": 25.106075566841827,
          "epoch_movement_energy_J": 0.073034299138466,
          "dally_2023_all_charged_reads_small_RAM_same_schedule_sensitivity_J": 1.236645855138466,
          "coordinate_descent_sweeps": 2,
          "coordinate_descent_starts": 3,
          "winning_start": "independent_layer_optima",
          "tiles_by_invocation_rows": {
            "64": [
              {
                "layer": 0,
                "input_width": 784,
                "output_width": 2500,
                "tile_i": 32,
                "tile_j": 125
              },
              {
                "layer": 1,
                "input_width": 2500,
                "output_width": 2000,
                "tile_i": 64,
                "tile_j": 20
              },
              {
                "layer": 2,
                "input_width": 2000,
                "output_width": 1500,
                "tile_i": 64,
                "tile_j": 30
              },
              {
                "layer": 3,
                "input_width": 1500,
                "output_width": 1000,
                "tile_i": 64,
                "tile_j": 40
              },
              {
                "layer": 4,
                "input_width": 1000,
                "output_width": 500,
                "tile_i": 64,
                "tile_j": 50
              },
              {
                "layer": 5,
                "input_width": 500,
                "output_width": 10,
                "tile_i": 64,
                "tile_j": 10
              }
            ],
            "32": [
              {
                "layer": 0,
                "input_width": 784,
                "output_width": 2500,
                "tile_i": 32,
                "tile_j": 125
              },
              {
                "layer": 1,
                "input_width": 2500,
                "output_width": 2000,
                "tile_i": 32,
                "tile_j": 25
              },
              {
                "layer": 2,
                "input_width": 2000,
                "output_width": 1500,
                "tile_i": 32,
                "tile_j": 30
              },
              {
                "layer": 3,
                "input_width": 1500,
                "output_width": 1000,
                "tile_i": 32,
                "tile_j": 40
              },
              {
                "layer": 4,
                "input_width": 1000,
                "output_width": 500,
                "tile_i": 32,
                "tile_j": 50
              },
              {
                "layer": 5,
                "input_width": 500,
                "output_width": 10,
                "tile_i": 32,
                "tile_j": 10
              }
            ]
          },
          "movement_energy_ratio_to_full_batch": 0.547061699773546
        },
        "16": {
          "batch_size": 16,
          "batch_invocations": 3750,
          "decomposition": [
            {
              "rows": 16,
              "invocation_count": 3750
            }
          ],
          "epoch_charged_reads": 2943519150000,
          "movement_intensity_grid_steps_per_charged_read": 45.6289253605498,
          "epoch_movement_energy_J": 0.134309615592699,
          "dally_2023_all_charged_reads_small_RAM_same_schedule_sensitivity_J": 1.3117172755926991,
          "coordinate_descent_sweeps": 3,
          "coordinate_descent_starts": 3,
          "winning_start": "all_minimum_tiles",
          "tiles_by_invocation_rows": {
            "16": [
              {
                "layer": 0,
                "input_width": 784,
                "output_width": 2500,
                "tile_i": 16,
                "tile_j": 250
              },
              {
                "layer": 1,
                "input_width": 2500,
                "output_width": 2000,
                "tile_i": 16,
                "tile_j": 20
              },
              {
                "layer": 2,
                "input_width": 2000,
                "output_width": 1500,
                "tile_i": 16,
                "tile_j": 25
              },
              {
                "layer": 3,
                "input_width": 1500,
                "output_width": 1000,
                "tile_i": 16,
                "tile_j": 25
              },
              {
                "layer": 4,
                "input_width": 1000,
                "output_width": 500,
                "tile_i": 16,
                "tile_j": 25
              },
              {
                "layer": 5,
                "input_width": 500,
                "output_width": 10,
                "tile_i": 16,
                "tile_j": 10
              }
            ]
          },
          "movement_energy_ratio_to_full_batch": 1.0060430163473015
        },
        "4": {
          "batch_size": 4,
          "batch_invocations": 15000,
          "decomposition": [
            {
              "rows": 4,
              "invocation_count": 15000
            }
          ],
          "epoch_charged_reads": 3081125400000,
          "movement_intensity_grid_steps_per_charged_read": 141.39913465321374,
          "epoch_movement_energy_J": 0.435668465318037,
          "dally_2023_all_charged_reads_small_RAM_same_schedule_sensitivity_J": 1.6681186253180371,
          "coordinate_descent_sweeps": 3,
          "coordinate_descent_starts": 3,
          "winning_start": "independent_layer_optima",
          "tiles_by_invocation_rows": {
            "4": [
              {
                "layer": 0,
                "input_width": 784,
                "output_width": 2500,
                "tile_i": 4,
                "tile_j": 250
              },
              {
                "layer": 1,
                "input_width": 2500,
                "output_width": 2000,
                "tile_i": 4,
                "tile_j": 20
              },
              {
                "layer": 2,
                "input_width": 2000,
                "output_width": 1500,
                "tile_i": 4,
                "tile_j": 20
              },
              {
                "layer": 3,
                "input_width": 1500,
                "output_width": 1000,
                "tile_i": 4,
                "tile_j": 20
              },
              {
                "layer": 4,
                "input_width": 1000,
                "output_width": 500,
                "tile_i": 4,
                "tile_j": 20
              },
              {
                "layer": 5,
                "input_width": 500,
                "output_width": 10,
                "tile_i": 4,
                "tile_j": 10
              }
            ]
          },
          "movement_energy_ratio_to_full_batch": 3.263364391609379
        },
        "1": {
          "batch_size": 1,
          "batch_invocations": 60000,
          "decomposition": [
            {
              "rows": 1,
              "invocation_count": 60000
            }
          ],
          "epoch_charged_reads": 3626300400000,
          "movement_intensity_grid_steps_per_charged_read": 461.51661484542456,
          "epoch_movement_energy_J": 1.673597885020609,
          "dally_2023_all_charged_reads_small_RAM_same_schedule_sensitivity_J": 3.1241180450206087,
          "coordinate_descent_sweeps": 3,
          "coordinate_descent_starts": 3,
          "winning_start": "independent_layer_optima",
          "tiles_by_invocation_rows": {
            "1": [
              {
                "layer": 0,
                "input_width": 784,
                "output_width": 2500,
                "tile_i": 1,
                "tile_j": 250
              },
              {
                "layer": 1,
                "input_width": 2500,
                "output_width": 2000,
                "tile_i": 1,
                "tile_j": 16
              },
              {
                "layer": 2,
                "input_width": 2000,
                "output_width": 1500,
                "tile_i": 1,
                "tile_j": 15
              },
              {
                "layer": 3,
                "input_width": 1500,
                "output_width": 1000,
                "tile_i": 1,
                "tile_j": 20
              },
              {
                "layer": 4,
                "input_width": 1000,
                "output_width": 500,
                "tile_i": 1,
                "tile_j": 20
              },
              {
                "layer": 5,
                "input_width": 500,
                "output_width": 10,
                "tile_i": 1,
                "tile_j": 10
              }
            ]
          },
          "movement_energy_ratio_to_full_batch": 12.536045591140265
        }
      }
    },
    "dally_2023_endpoint_sensitivity": {
      "formula": "E_J = movement_energy_J + charged_reads * 400e-15 for INT8",
      "source": "https://aha.stanford.edu/sites/g/files/sbiybj20066/files/media/file/aha-retreat-2023_dally_keynote_en_eff_ai_hw_0.pdf",
      "source_values": "100 fJ/(bit mm) on-chip communication and 50 fJ/bit small-RAM access",
      "qualification": "illustrative sensitivity only: it treats every abstract charged read, including scratch and final-output reads, as an 8-bit small-RAM access and evaluates the movement-optimized tiles without retuning; it is not part of the movement-only model"
    }
  },
  "mac_count_only_sensitivity": {
    "reference": "retuned 8192^3 sa_cache movement energy",
    "reference_MAC": 549755813888,
    "reference_energy_J": 0.118953083721334,
    "energy_J_per_MAC": 2.1637439880820965e-13,
    "projected_MNIST_epoch_energy_J_for_either_batching": 0.1553351809044137
  },
  "precision_semantics": "The model has unbounded scalar cells and no bit width; its movement energy does not change between FP32, FP16, INT8, or one-bit values.",
  "omitted_terms": [
    "destination writes",
    "arithmetic, bias, ReLU, saturation, and quantization",
    "Tensor Core shape padding",
    "energy to initially materialize the resident dataset and any batch-gather or copy operation",
    "kernel launch, control, clocking, leakage, and elapsed time",
    "physical cache, SRAM, HBM, and host-transfer behavior",
    "GPU occupancy and underfilled-kernel efficiency"
  ]
}
