{
  "report_date": "2026-08-28",
  "workload": {
    "m": 8192,
    "n": 8192,
    "k": 8192,
    "macs": 549755813888,
    "conventional_operations": 1099511627776,
    "vendor_counting_convention": "one multiply-accumulate equals two operations",
    "comparison_contract": "INT8 inputs with INT32 accumulation/output unless a row says otherwise",
    "input_bytes": 134217728,
    "int32_output_bytes": 268435456,
    "minimum_logical_device_traffic_bytes": 402653184
  },
  "sutro_model": {
    "energy_coefficient_J_per_score_unit": 1e-15,
    "score_unit": "one source operand times one abstract address-grid step",
    "omitted_terms": [
      "destination writes",
      "arithmetic",
      "memory capacity and physical hierarchy",
      "HBM transfers",
      "control and clocking",
      "leakage and time"
    ],
    "schedules": [
      {
        "name": "current record cubically scaled lower bound",
        "score": 8898635366400,
        "energy_J": 0.0088986353664,
        "qualification": "not a generated 8K program; assumes fixed average distance"
      },
      {
        "name": "retuned sa_cache",
        "tile_i": 256,
        "tile_j": 128,
        "tile_candidates_evaluated": 196,
        "search_scope": "all ordered pairs of the 14 divisors of 8192 within the literal generalized sa_cache family",
        "validation": "the same formula reproduces sa_cache(16,8,4)=73602 and naive(16)=340704",
        "paid_reads": 2205465706496,
        "score": 118953083721334,
        "energy_J": 0.118953083721334,
        "fJ_per_conventional_operation": 108.187199404105,
        "movement_only_TOPS_per_W_equivalent": 9.24323769825,
        "improvement_over_naive": 196.727009216
      },
      {
        "name": "retuned square tiled",
        "tile": 64,
        "score": 317312291631114,
        "energy_J": 0.317312291631114
      },
      {
        "name": "fixed 4x4 tiled",
        "score": 2136290066409474,
        "energy_J": 2.136290066409474
      },
      {
        "name": "naive baseline",
        "score": 23401284397481984,
        "energy_J": 23.401284397481984
      }
    ],
    "retuned_sa_cache_derivation": {
      "distance": "d(a)=ceil(sqrt(a))",
      "range_sum": "D(l,h)=F(h)-F(l-1)",
      "prefix_sum": "F(x)=q(q+1)(4q-1)/6+(x-q^2)(q+1), q=floor(sqrt(x))",
      "score_formula": "N^3*d(1)+N^2*(N-1)*d(2)+(N^3/Tj)*D(sB)+(N^3/(Ti*Tj))*D(sC)+(N/Tj)*D(A)+(N/Ti)*D(B)+D(C)",
      "address_regions": {
        "sA": [1, 1],
        "TMP": [2, 2],
        "sB": [3, 130],
        "sC": [131, 32898],
        "A": [32899, 67141762],
        "B": [67141763, 134250626],
        "C": [134250627, 201359490]
      },
      "reads_per_cell": {
        "sA": 549755813888,
        "TMP": 549688705024,
        "sB": 4294967296,
        "sC": 16777216,
        "A": 64,
        "B": 32,
        "C": 1
      },
      "score_units_by_region": {
        "sA": 549755813888,
        "TMP": 1099377410048,
        "sB": 4514010628096,
        "sC": 66997983379456,
        "A": 23475391008000,
        "B": 21448665777152,
        "C": 867899704694,
        "total": 118953083721334
      },
      "next_tile_candidates": [
        {"tile_i": 128, "tile_j": 128, "score": 121105737911286},
        {"tile_i": 256, "tile_j": 64, "score": 121681290169654}
      ],
      "qualification": "exact closed-form evaluation for this schedule family; not a proof over all algorithms and no 8K IR is materialized"
    },
    "retuned_sa_cache_breakdown_J": {
      "bulk_A_loads": 0.0234753910,
      "bulk_B_loads": 0.0214486658,
      "C_exit": 0.0008678997,
      "near_ALU_A_stage": 0.0005497558,
      "B_stage": 0.0045140106,
      "accumulator_stage": 0.0010993774,
      "output_tile_stage": 0.0669979834
    },
    "endpoint_sensitivity": {
      "formula": "2.5e-15 * (S + 160*R)",
      "S": 118953083721334,
      "R": 2205465706496,
      "energy_J": 1.179568991901735
    }
  },
  "nvidia_estimates": [
    {
      "sku": "B200 Blackwell",
      "dense_INT8_TOPS": 4500,
      "power_W": 1000,
      "power_scope": "maximum configurable TDP",
      "ideal_time_s": 0.0002443359172835556,
      "energy_J": 0.2443359172835556,
      "source": "https://dam-cdn.nvd.orangelogic.com/AssetLink/y441155802qub41q118b2852i557jem5.pdf",
      "evidence": "peak-throughput/TDP estimate; not measured"
    },
    {
      "sku": "B300 Blackwell Ultra",
      "dense_INT8_TOPS": 153.5,
      "power_W": 1100,
      "power_scope": "maximum configurable TDP",
      "ideal_time_s": 0.007162942200495114,
      "energy_J": 7.879236420544625,
      "source": "https://dam-cdn.nvd.orangelogic.com/AssetLink/1k0p832eq8r5ca0u5383ie5o4tp3bst1.pdf",
      "evidence": "peak-throughput/TDP estimate; not measured"
    },
    {
      "sku": "Rubin",
      "dense_INT8_TOPS": 250,
      "power_W": 1800,
      "power_scope": "datasheet application-comparison assumption, not formally labeled TDP",
      "ideal_time_s": 0.004398046511104,
      "energy_J": 7.916483719987201,
      "source": "https://www.nvidia.com/en-us/data-center/vera-rubin-nvl72/",
      "power_source": "https://dam-cdn.nvd.orangelogic.com/AssetLink/v5rf2icnf86o26e464tf6djn23r8ibhe.pdf",
      "evidence": "preliminary peak-throughput/assumed-power estimate; not measured"
    }
  ],
  "blackwell_b200_precision_estimates": [
    {
      "precision": "NVFP4",
      "stored_bits_per_input_element": 4,
      "dense_throughput_TOPS": 9000,
      "power_W": 1000,
      "ideal_time_s": 0.0001221679586417778,
      "energy_J": 0.1221679586417778,
      "minimum_input_bytes_excluding_scales_and_output": 67108864,
      "semantics": "block-scaled four-bit inputs; conversion, scale metadata, and higher-precision output are omitted"
    },
    {
      "precision": "INT8",
      "stored_bits_per_input_element": 8,
      "dense_throughput_TOPS": 4500,
      "power_W": 1000,
      "ideal_time_s": 0.0002443359172835556,
      "energy_J": 0.2443359172835556,
      "minimum_logical_bytes_with_INT32_output": 402653184
    },
    {
      "precision": "TF32",
      "stored_bits_per_input_element": 32,
      "dense_throughput_TOPS": 1100,
      "power_W": 1000,
      "ideal_time_s": 0.0009995560252509091,
      "energy_J": 0.9995560252509091,
      "minimum_logical_bytes_with_FP32_output": 805306368,
      "semantics": "FP32 storage and accumulation with reduced-precision TF32 multiplication"
    },
    {
      "precision": "FP32",
      "stored_bits_per_input_element": 32,
      "dense_throughput_TOPS": 75,
      "power_W": 1000,
      "ideal_time_s": 0.014660155037013333,
      "energy_J": 14.660155037013332,
      "minimum_logical_bytes_with_FP32_output": 805306368,
      "semantics": "literal four-byte FP32 path"
    }
  ],
  "blackwell_b300_precision_estimates": [
    {
      "precision": "NVFP4",
      "stored_bits_per_input_element": 4,
      "dense_throughput_TOPS": 14000,
      "power_W": 1100,
      "ideal_time_s": 0.00007853654484114286,
      "energy_J": 0.08639019932525714,
      "qualification": "individual-GPU table rounds dense throughput to 14 PFLOPS; 8-GPU aggregate implies 13.5 PFLOPS per GPU and 0.0895898363373 J"
    },
    {
      "precision": "TF32",
      "stored_bits_per_input_element": 32,
      "dense_throughput_TOPS": 1100,
      "power_W": 1100,
      "ideal_time_s": 0.0009995560252509091,
      "energy_J": 1.0995116277760002,
      "semantics": "FP32 storage and accumulation with reduced-precision TF32 multiplication"
    },
    {
      "precision": "FP32",
      "stored_bits_per_input_element": 32,
      "dense_throughput_TOPS": 75,
      "power_W": 1100,
      "ideal_time_s": 0.014660155037013333,
      "energy_J": 16.126170540714664,
      "semantics": "literal four-byte FP32 path"
    }
  ],
  "formula_for_peak_power_rows": "energy_J = conventional_operations / (dense_TOPS * 1e12) * power_W",
  "semantic_warning": "The repository verifier uses unbounded Python integers; the score is value-independent, but the INT8 table is not semantic equivalence."
}
