{
  "generated_at": "2026-07-30T23:39:09Z",
  "pipeline_version": "2.0.0",
  "sources": [
    {
      "id": "S4",
      "name": "Epoch AI, Epoch Capabilities Index benchmark scores",
      "url": "https://epoch.ai/data/eci_benchmarks.csv",
      "retrieved_at": "2026-07-30T23:22:50Z",
      "sha256": "d7cad7a8595347a62a2f832205aae579e371f05afe6ab08bc9506631e38c70d1",
      "bytes": 306562,
      "licence": "CC-BY 4.0",
      "attribution": "Epoch AI, Epoch Capabilities Index benchmark data, CC-BY 4.0, https://epoch.ai/eci",
      "cache_file": "cache/eci_benchmarks.csv",
      "fetch_note": "cache only (--offline)",
      "staleness_days": 0
    },
    {
      "id": "S5",
      "name": "METR, task time horizon benchmark results v1.1",
      "url": "https://metr.org/assets/benchmark_results_1_1.yaml",
      "retrieved_at": "2026-07-30T23:22:51Z",
      "sha256": "aae31902b0519a4da73e16643915e5e8aca13cd3315c3aac893ce3d6dfe92ad9",
      "bytes": 16255,
      "licence": "not stated by the publisher; attribution given as a matter of practice",
      "attribution": "METR, Measuring AI Ability to Complete Long Tasks, benchmark_results_1_1.yaml, https://metr.org/time-horizons/",
      "cache_file": "cache/metr_benchmark_results_1_1.yaml",
      "fetch_note": "cache only (--offline)",
      "staleness_days": 108
    },
    {
      "id": "S2",
      "name": "Epoch AI, Notable AI Models",
      "url": "https://epoch.ai/data/notable_ai_models.csv",
      "retrieved_at": "2026-07-30T23:22:51Z",
      "sha256": "a38f2ef6ea1ffbcc966e3bad862375332046afbe4e0b690f6143ae65329f1495",
      "bytes": 2207395,
      "licence": "CC-BY 4.0",
      "attribution": "Epoch AI, Notable AI Models, CC-BY 4.0, https://epoch.ai/data/notable-ai-models",
      "cache_file": "cache/notable_ai_models.csv",
      "fetch_note": "cache only (--offline)",
      "staleness_days": 0
    },
    {
      "id": "S3",
      "name": "Epoch AI, Machine Learning Hardware",
      "url": "https://epoch.ai/data/ml_hardware.csv",
      "retrieved_at": "2026-07-30T23:22:51Z",
      "sha256": "9ecfe145b60360cf4bb66399106ee6a7c57b1c2e98668cae558b0ede16e96e91",
      "bytes": 97233,
      "licence": "CC-BY 4.0",
      "attribution": "Epoch AI, Machine Learning Hardware, CC-BY 4.0, https://epoch.ai/data/machine-learning-hardware",
      "cache_file": "cache/ml_hardware.csv",
      "fetch_note": "cache only (--offline)",
      "staleness_days": 135
    }
  ],
  "headline": {
    "position": {
      "met": 1,
      "countable": 4,
      "registered": 4,
      "low": 1,
      "high": 3,
      "method": "Count of pre-registered thresholds met, in native units, never a normalised level. Four thresholds are registered at pipeline version 2.0.0: two on A1 headroom and two on the METR task time horizons. 4 of the 4 could be evaluated against real data in this run. The low and high bounds come from comparing each indicator's published confidence interval against its threshold, so a threshold counts in `high` when the optimistic end of the interval clears it. Thresholds and their per-threshold states are enumerated in notes.thresholds.",
      "state": "PROVISIONAL"
    },
    "rate": {
      "value": 128.744,
      "unit": "days per doubling of the task time horizon",
      "low": 104.428,
      "high": 158.012,
      "n": 26,
      "method": "METR's published doubling time for observations from 2023 on, with METR's own confidence interval. n is the count of model results present in the fetched METR file, not the count of points inside METR's fit, which the file does not enumerate. The file does not state which horizon threshold the fit is taken on; its only comment on the field says the fit excludes points whose central p50 estimate exceeds 16 hours, so the unit here is left as the task time horizon rather than asserting p50 or p80.",
      "source_id": "S5",
      "verification_class": "independently-verified"
    },
    "flag": {
      "primary": "STALE",
      "conditions": [
        "STALE(S5:108d)",
        "DISPERSION-GAP(B,D,E,F)"
      ]
    },
    "floor": {
      "dimension": "C",
      "met": 0,
      "of": 2,
      "k": 2,
      "core_total": 4,
      "note": "Weakest link across the 2 core dimensions that carry thresholds, out of 4 core dimensions in total. Dimensions B and D set no thresholds, so this floor is an upper bound: adding a dimension to a minimum can only lower it."
    }
  },
  "indicators": [
    {
      "id": "A1",
      "name": "Frontier Eval Headroom",
      "dimension": "A",
      "class": "core",
      "value": 0.3895,
      "unit": "headroom (0 to 1)",
      "display": "0.390",
      "status": "measured",
      "direction": "flat",
      "polarity": "good",
      "uncertainty": null,
      "series": [
        {
          "date": "2024-05-31",
          "value": 0.9
        },
        {
          "date": "2024-06-30",
          "value": 0.9005
        },
        {
          "date": "2024-07-31",
          "value": 0.9005
        },
        {
          "date": "2024-08-31",
          "value": 0.9005
        },
        {
          "date": "2024-09-30",
          "value": 0.9005
        },
        {
          "date": "2024-10-31",
          "value": 0.9
        },
        {
          "date": "2024-11-30",
          "value": 0.9
        },
        {
          "date": "2024-12-31",
          "value": 0.674
        },
        {
          "date": "2025-01-31",
          "value": 0.706
        },
        {
          "date": "2025-02-28",
          "value": 0.6915
        },
        {
          "date": "2025-03-31",
          "value": 0.6915
        },
        {
          "date": "2025-04-30",
          "value": 0.6554
        },
        {
          "date": "2025-05-31",
          "value": 0.6538
        },
        {
          "date": "2025-06-30",
          "value": 0.6538
        },
        {
          "date": "2025-07-31",
          "value": 0.6469
        },
        {
          "date": "2025-08-31",
          "value": 0.564
        },
        {
          "date": "2025-09-30",
          "value": 0.571
        },
        {
          "date": "2025-10-31",
          "value": 0.571
        },
        {
          "date": "2025-11-30",
          "value": 0.545
        },
        {
          "date": "2025-12-31",
          "value": 0.503
        },
        {
          "date": "2026-01-31",
          "value": 0.503
        },
        {
          "date": "2026-02-28",
          "value": 0.503
        },
        {
          "date": "2026-03-31",
          "value": 0.4469
        },
        {
          "date": "2026-04-30",
          "value": 0.4329
        },
        {
          "date": "2026-05-31",
          "value": 0.3895
        },
        {
          "date": "2026-06-30",
          "value": 0.3895
        },
        {
          "date": "2026-07-31",
          "value": 0.3895
        }
      ],
      "method": "Median of (1 minus best score achieved to date) across a basket of 28 benchmarks from the Epoch ECI file that are unsaturated (no model has yet reached 0.90), carry at least 10 scored models, and received at least one score in the 365 days before the data as-of date; the monthly series starts at the first month in which 8 basket members coexist.",
      "source_id": "S4",
      "verification_class": "inferred",
      "observed_at": "2026-07-24",
      "note": "Basket membership and per-benchmark headroom are enumerated in notes.a1_basket. Uncertainty is not stated because the Epoch file publishes point scores with no interval.",
      "provenance_class": "derived",
      "provenance_note": "Computed by this tracker from published raw rows. No upstream source publishes this figure."
    },
    {
      "id": "A2",
      "name": "Benchmark Saturation Half-Life",
      "dimension": "A",
      "class": "core",
      "value": 20.83,
      "unit": "months",
      "display": "20.8 months",
      "status": "measured",
      "direction": "flat",
      "polarity": "context",
      "uncertainty": null,
      "series": [
        {
          "date": "2023-03-31",
          "value": 31.21
        },
        {
          "date": "2023-04-30",
          "value": 31.21
        },
        {
          "date": "2023-05-31",
          "value": 31.21
        },
        {
          "date": "2023-06-30",
          "value": 31.21
        },
        {
          "date": "2023-07-31",
          "value": 31.21
        },
        {
          "date": "2023-08-31",
          "value": 31.21
        },
        {
          "date": "2023-09-30",
          "value": 31.21
        },
        {
          "date": "2023-10-31",
          "value": 31.21
        },
        {
          "date": "2023-11-30",
          "value": 31.21
        },
        {
          "date": "2023-12-31",
          "value": 31.21
        },
        {
          "date": "2024-01-31",
          "value": 31.21
        },
        {
          "date": "2024-02-29",
          "value": 31.21
        },
        {
          "date": "2024-03-31",
          "value": 31.21
        },
        {
          "date": "2024-04-30",
          "value": 31.21
        },
        {
          "date": "2024-05-31",
          "value": 31.21
        },
        {
          "date": "2024-06-30",
          "value": 31.21
        },
        {
          "date": "2024-07-31",
          "value": 45.87
        },
        {
          "date": "2024-08-31",
          "value": 45.87
        },
        {
          "date": "2024-09-30",
          "value": 45.87
        },
        {
          "date": "2024-10-31",
          "value": 45.87
        },
        {
          "date": "2024-11-30",
          "value": 45.87
        },
        {
          "date": "2024-12-31",
          "value": 45.45
        },
        {
          "date": "2025-01-31",
          "value": 45.45
        },
        {
          "date": "2025-02-28",
          "value": 45.45
        },
        {
          "date": "2025-03-31",
          "value": 45.45
        },
        {
          "date": "2025-04-30",
          "value": 45.45
        },
        {
          "date": "2025-05-31",
          "value": 45.45
        },
        {
          "date": "2025-06-30",
          "value": 45.04
        },
        {
          "date": "2025-07-31",
          "value": 45.04
        },
        {
          "date": "2025-08-31",
          "value": 30.8
        },
        {
          "date": "2025-09-30",
          "value": 30.8
        },
        {
          "date": "2025-10-31",
          "value": 30.8
        },
        {
          "date": "2025-11-30",
          "value": 23.95
        },
        {
          "date": "2025-12-31",
          "value": 34.49
        },
        {
          "date": "2026-01-31",
          "value": 34.49
        },
        {
          "date": "2026-02-28",
          "value": 23.95
        },
        {
          "date": "2026-03-31",
          "value": 23.95
        },
        {
          "date": "2026-04-30",
          "value": 20.83
        },
        {
          "date": "2026-05-31",
          "value": 20.83
        },
        {
          "date": "2026-06-30",
          "value": 20.83
        },
        {
          "date": "2026-07-31",
          "value": 20.83
        }
      ],
      "method": "For every benchmark in the Epoch ECI file that states a release date, the months from that release date to the first observation at or above 90 percent of the ceiling of 1.0; the reported value is the median over the 10 benchmarks that crossed, while 23 benchmarks with a release date have not crossed and 21 benchmarks state no release date and are excluded.",
      "source_id": "S4",
      "verification_class": "inferred",
      "observed_at": "2026-07-24",
      "note": "The Epoch `date` column is the model date, not the date the score was run, so a crossing is dated to the model that achieved it. This measures the instrument failing, not the systems: a short half-life means the rulers stop discriminating quickly. Crossings and non-crossings are enumerated in notes.a2_crossings and notes.a2_uncrossed.",
      "provenance_class": "derived",
      "provenance_note": "Computed by this tracker from published raw rows. No upstream source publishes this figure."
    },
    {
      "id": "B1",
      "name": "Held-Out Novelty Gap",
      "dimension": "B",
      "class": "core",
      "value": null,
      "unit": null,
      "display": null,
      "status": "not_measured",
      "direction": "flat",
      "polarity": "context",
      "uncertainty": null,
      "series": [],
      "method": "Not computed. No verified source in this run supplies the required input.",
      "source_id": null,
      "verification_class": "missing",
      "observed_at": null,
      "note": "No fetched source publishes matched public and withheld scores for one model. ARC Prize and SWE-bench Pro leaderboards are HTML only and publish no machine-readable withheld split.",
      "provenance_class": null
    },
    {
      "id": "B2",
      "name": "Contamination-Controlled Score Ratio",
      "dimension": "B",
      "class": "core",
      "value": null,
      "unit": null,
      "display": null,
      "status": "not_measured",
      "direction": "flat",
      "polarity": "context",
      "uncertainty": null,
      "series": [],
      "method": "Not computed. No verified source in this run supplies the required input.",
      "source_id": null,
      "verification_class": "missing",
      "observed_at": null,
      "note": "Defined on B1's qualifying pairs. B1 has no data, so B2 has no denominator.",
      "provenance_class": null
    },
    {
      "id": "C1",
      "name": "50% Task Time Horizon",
      "dimension": "C",
      "class": "core",
      "value": 1044.7801,
      "unit": "minutes of human expert task time",
      "display": "1045 min",
      "status": "measured",
      "direction": "up",
      "polarity": "good",
      "uncertainty": {
        "low": 508.8768,
        "high": 3304.2612
      },
      "series": [
        {
          "date": "2019-02-14",
          "value": 0.0538
        },
        {
          "date": "2020-05-28",
          "value": 0.1441
        },
        {
          "date": "2022-03-15",
          "value": 0.5992
        },
        {
          "date": "2023-03-14",
          "value": 3.9874
        },
        {
          "date": "2023-11-06",
          "value": 4.045
        },
        {
          "date": "2024-03-04",
          "value": 3.9523
        },
        {
          "date": "2024-04-09",
          "value": 3.7328
        },
        {
          "date": "2024-05-13",
          "value": 6.9912
        },
        {
          "date": "2024-06-20",
          "value": 11.3954
        },
        {
          "date": "2024-09-12",
          "value": 20.3266
        },
        {
          "date": "2024-10-22",
          "value": 20.5229
        },
        {
          "date": "2024-12-05",
          "value": 38.8316
        },
        {
          "date": "2025-02-24",
          "value": 60.3889
        },
        {
          "date": "2025-04-16",
          "value": 119.7326
        },
        {
          "date": "2025-05-22",
          "value": 100.3661
        },
        {
          "date": "2025-08-05",
          "value": 100.472
        },
        {
          "date": "2025-08-07",
          "value": 203.0126
        },
        {
          "date": "2025-11-18",
          "value": 224.3259
        },
        {
          "date": "2025-11-19",
          "value": 223.7147
        },
        {
          "date": "2025-11-24",
          "value": 292.9946
        },
        {
          "date": "2025-12-11",
          "value": 352.2493
        },
        {
          "date": "2026-02-05",
          "value": 718.8068
        },
        {
          "date": "2026-02-05",
          "value": 349.5307
        },
        {
          "date": "2026-02-19",
          "value": 384.1474
        },
        {
          "date": "2026-03-05",
          "value": 341.7353
        },
        {
          "date": "2026-04-07",
          "value": 1044.7801
        }
      ],
      "method": "The p50 horizon length published by METR for the most recent model its file flags `is_sota`, read from benchmark_results_1_1.yaml with a purpose-written indentation parser rather than pyyaml, which is not installed.",
      "source_id": "S5",
      "verification_class": "independently-verified",
      "observed_at": "2026-04-07",
      "note": "METR runs these evaluations itself, so the class is independently-verified; the source register nevertheless rates METR MEDIUM auditability because its funding, its pre-deployment access to the labs it times, and the log-linear form of its fit have not been examined. Model name: claude_mythos_preview_early_inspect.",
      "provenance_class": "restated",
      "provenance_note": "Restates a figure the publisher reports itself."
    },
    {
      "id": "C2",
      "name": "80% Task Time Horizon",
      "dimension": "C",
      "class": "core",
      "value": 185.9118,
      "unit": "minutes of human expert task time",
      "display": "186 min",
      "status": "measured",
      "direction": "up",
      "polarity": "good",
      "uncertainty": {
        "low": 97.3029,
        "high": 398.5146
      },
      "series": [
        {
          "date": "2019-02-14",
          "value": 0.0128
        },
        {
          "date": "2020-05-28",
          "value": 0.0562
        },
        {
          "date": "2022-03-15",
          "value": 0.2554
        },
        {
          "date": "2023-03-14",
          "value": 0.8896
        },
        {
          "date": "2023-11-06",
          "value": 0.783
        },
        {
          "date": "2024-03-04",
          "value": 0.639
        },
        {
          "date": "2024-04-09",
          "value": 0.9279
        },
        {
          "date": "2024-05-13",
          "value": 1.267
        },
        {
          "date": "2024-06-20",
          "value": 1.6718
        },
        {
          "date": "2024-09-12",
          "value": 4.4205
        },
        {
          "date": "2024-10-22",
          "value": 2.5957
        },
        {
          "date": "2024-12-05",
          "value": 7.0901
        },
        {
          "date": "2025-02-24",
          "value": 12.0918
        },
        {
          "date": "2025-04-16",
          "value": 29.9816
        },
        {
          "date": "2025-05-22",
          "value": 20.4298
        },
        {
          "date": "2025-08-05",
          "value": 23.4558
        },
        {
          "date": "2025-08-07",
          "value": 38.3124
        },
        {
          "date": "2025-11-18",
          "value": 54.1428
        },
        {
          "date": "2025-11-19",
          "value": 50.6325
        },
        {
          "date": "2025-11-24",
          "value": 49.4306
        },
        {
          "date": "2025-12-11",
          "value": 66.0026
        },
        {
          "date": "2026-02-05",
          "value": 69.8746
        },
        {
          "date": "2026-02-05",
          "value": 54.7394
        },
        {
          "date": "2026-02-19",
          "value": 89.8015
        },
        {
          "date": "2026-03-05",
          "value": 53.8779
        },
        {
          "date": "2026-04-07",
          "value": 185.9118
        }
      ],
      "method": "The p80 horizon length published by METR for the same most recent SOTA-flagged model as C1, parsed from the same file.",
      "source_id": "S5",
      "verification_class": "independently-verified",
      "observed_at": "2026-04-07",
      "note": "The 80 percent threshold is the stricter reliability bar and is the one carrying a headline threshold.",
      "provenance_class": "restated",
      "provenance_note": "Restates a figure the publisher reports itself."
    },
    {
      "id": "C3",
      "name": "Horizon Doubling Time",
      "dimension": "C",
      "class": "core",
      "value": 128.744,
      "unit": "days per doubling",
      "display": "129 days",
      "status": "measured",
      "direction": "down",
      "polarity": "context",
      "uncertainty": {
        "low": 104.428,
        "high": 158.012
      },
      "series": [
        {
          "date": "2019-02-14",
          "value": 187.778
        },
        {
          "date": "2026-04-07",
          "value": 128.744
        }
      ],
      "method": "METR's published `doubling_time_in_days.from_2023_on` point estimate with its stated confidence interval, read straight from the top level of benchmark_results_1_1.yaml; the earlier series point is METR's `all_time_stitched` estimate over the full history, so the two points show the fit steepening, not a monthly trend.",
      "source_id": "S5",
      "verification_class": "independently-verified",
      "observed_at": "2026-04-07",
      "note": "This is METR's own fit, restated, not a tracker computation. The file states the fit excludes points whose central p50 estimate exceeds 16 hours. Series carries two points only, so it is a comparison of two published fits and never a trend line. All-time stitched estimate: 187.778 days.",
      "provenance_class": "restated",
      "provenance_note": "Restates a figure the publisher reports itself."
    },
    {
      "id": "D1",
      "name": "pass@1 to pass@k Spread",
      "dimension": "D",
      "class": "core",
      "value": null,
      "unit": null,
      "display": null,
      "status": "not_measured",
      "direction": "flat",
      "polarity": "context",
      "uncertainty": null,
      "series": [],
      "method": "Not computed. No verified source in this run supplies the required input.",
      "source_id": null,
      "verification_class": "missing",
      "observed_at": null,
      "note": "Requires per-attempt results. The fetched sources carry one aggregate score per model per benchmark and no attempt-level records.",
      "provenance_class": null
    },
    {
      "id": "D2",
      "name": "Rerun Variance",
      "dimension": "D",
      "class": "core",
      "value": null,
      "unit": null,
      "display": null,
      "status": "not_measured",
      "direction": "flat",
      "polarity": "context",
      "uncertainty": null,
      "series": [],
      "method": "Not computed. No verified source in this run supplies the required input.",
      "source_id": null,
      "verification_class": "missing",
      "observed_at": null,
      "note": "Requires the tracker to run its own repeated probes against model endpoints. No inference budget and no probe harness exist, and no fetched source publishes rerun dispersion.",
      "provenance_class": null
    },
    {
      "id": "D3",
      "name": "Calibration Error",
      "dimension": "D",
      "class": "core",
      "value": null,
      "unit": null,
      "display": null,
      "status": "not_measured",
      "direction": "flat",
      "polarity": "context",
      "uncertainty": null,
      "series": [],
      "method": "Not computed. No verified source in this run supplies the required input.",
      "source_id": null,
      "verification_class": "missing",
      "observed_at": null,
      "note": "Requires elicited per-item confidence. No fetched source publishes calibration, and the one that used to (Stanford HELM) entered maintenance mode on 2026-06-01.",
      "provenance_class": null
    },
    {
      "id": "D4",
      "name": "Generation-over-Generation Regression Count",
      "dimension": "D",
      "class": "core",
      "value": 40,
      "unit": "regressions",
      "display": "40 of 611 comparable cells",
      "status": "measured",
      "direction": "flat",
      "polarity": "bad",
      "uncertainty": null,
      "series": [
        {
          "date": "2023-07-31",
          "value": 2
        },
        {
          "date": "2023-08-31",
          "value": 2
        },
        {
          "date": "2023-09-30",
          "value": 2
        },
        {
          "date": "2023-10-31",
          "value": 2
        },
        {
          "date": "2023-11-30",
          "value": 2
        },
        {
          "date": "2023-12-31",
          "value": 3
        },
        {
          "date": "2024-01-31",
          "value": 3
        },
        {
          "date": "2024-02-29",
          "value": 3
        },
        {
          "date": "2024-03-31",
          "value": 3
        },
        {
          "date": "2024-04-30",
          "value": 3
        },
        {
          "date": "2024-05-31",
          "value": 3
        },
        {
          "date": "2024-06-30",
          "value": 3
        },
        {
          "date": "2024-07-31",
          "value": 3
        },
        {
          "date": "2024-08-31",
          "value": 3
        },
        {
          "date": "2024-09-30",
          "value": 4
        },
        {
          "date": "2024-10-31",
          "value": 4
        },
        {
          "date": "2024-11-30",
          "value": 4
        },
        {
          "date": "2024-12-31",
          "value": 8
        },
        {
          "date": "2025-01-31",
          "value": 9
        },
        {
          "date": "2025-02-28",
          "value": 9
        },
        {
          "date": "2025-03-31",
          "value": 9
        },
        {
          "date": "2025-04-30",
          "value": 9
        },
        {
          "date": "2025-05-31",
          "value": 17
        },
        {
          "date": "2025-06-30",
          "value": 17
        },
        {
          "date": "2025-07-31",
          "value": 17
        },
        {
          "date": "2025-08-31",
          "value": 23
        },
        {
          "date": "2025-09-30",
          "value": 23
        },
        {
          "date": "2025-10-31",
          "value": 23
        },
        {
          "date": "2025-11-30",
          "value": 23
        },
        {
          "date": "2025-12-31",
          "value": 28
        },
        {
          "date": "2026-01-31",
          "value": 28
        },
        {
          "date": "2026-02-28",
          "value": 31
        },
        {
          "date": "2026-03-31",
          "value": 34
        },
        {
          "date": "2026-04-30",
          "value": 36
        },
        {
          "date": "2026-05-31",
          "value": 38
        },
        {
          "date": "2026-06-30",
          "value": 40
        },
        {
          "date": "2026-07-31",
          "value": 40
        }
      ],
      "method": "Caveat on this indicator's name: same-tier successor pairs include point releases inside a single version line, for example a March and a May snapshot of one model, so a minority of entries are within-generation revisions rather than generation-over-generation changes. Every entry is listed by name in notes.d4_regressions so a reader can tell which is which. Cumulative count of cases where a later model beats its predecessor overall yet scores lower on one specific benchmark. A pair qualifies only when both models carry the same inferred developer AND the same product-tier key (so a flagship is never compared against the mini or flash model that followed it), share at least 5 benchmarks whose benchmark_release_date matches exactly, and the later model's median delta across those shared benchmarks is positive; a cell counts only when the score falls by more than 0.02 absolute. 70 pairs were evaluated, 9 were rejected because the later model was not an overall improvement, and 17 models were excluded because no developer could be inferred from the name.",
      "source_id": "S4",
      "verification_class": "inferred",
      "observed_at": "2026-07-24",
      "note": "This is the tracker's only original measurement: no public source publishes it. Every regression is enumerated in notes.d4_regressions with both model names, the benchmark, both scores and both dates, so each entry is checkable against a row in the source file. Excluded by construction: cross-tier comparisons, cross-version benchmark comparisons, models whose developer could not be inferred, drops of 0.02 or less, and any later model that was not better overall. A false regression is worse than a missed one.",
      "provenance_class": "derived",
      "provenance_note": "Computed by this tracker from published raw rows. No upstream source publishes this figure."
    },
    {
      "id": "E1",
      "name": "Real-Task Win Rate vs Human Expert",
      "dimension": "E",
      "class": "supporting",
      "value": null,
      "unit": null,
      "display": null,
      "status": "not_measured",
      "direction": "flat",
      "polarity": "context",
      "uncertainty": null,
      "series": [],
      "method": "Not computed. No verified source in this run supplies the required input.",
      "source_id": null,
      "verification_class": "missing",
      "observed_at": null,
      "note": "The only candidate feed, Artificial Analysis, is proprietary and needs a written licence before any derived series may be published.",
      "provenance_class": null
    },
    {
      "id": "E2",
      "name": "Benchmark-to-Reality Gap",
      "dimension": "E",
      "class": "supporting",
      "value": null,
      "unit": null,
      "display": null,
      "status": "not_measured",
      "direction": "flat",
      "polarity": "context",
      "uncertainty": null,
      "series": [],
      "method": "Not computed. No verified source in this run supplies the required input.",
      "source_id": null,
      "verification_class": "missing",
      "observed_at": null,
      "note": "Structurally unquantifiable from any verified source. METR states publicly that it does not know the size of this gap, so any number here would be invented.",
      "provenance_class": null
    },
    {
      "id": "F1",
      "name": "Simulated Embodied-Task Success Rate",
      "dimension": "F",
      "class": "supporting",
      "value": null,
      "unit": null,
      "display": null,
      "status": "not_measured",
      "direction": "flat",
      "polarity": "context",
      "uncertainty": null,
      "series": [],
      "method": "Not computed. No verified source in this run supplies the required input.",
      "source_id": null,
      "verification_class": "missing",
      "observed_at": null,
      "note": "No robotics benchmark carries both unstructured-environment tasks and a machine-readable time series. The BEHAVIOR 2026 board is a hosted Gradio Space with no results file.",
      "provenance_class": null
    },
    {
      "id": "G1i",
      "name": "Compute to Reach a Fixed Capability Threshold",
      "dimension": "G",
      "class": "contextual",
      "value": null,
      "unit": null,
      "display": null,
      "status": "not_measured",
      "direction": "flat",
      "polarity": "context",
      "uncertainty": null,
      "series": [],
      "method": "Not computed. No verified source in this run supplies the required input.",
      "source_id": null,
      "verification_class": "missing",
      "observed_at": null,
      "note": "Needs a score-to-compute join. Only 86 of the 213 models scored in the ECI file appear by exact name in Notable AI Models, so the joined sample would be a biased subset and the trend would not be defensible.",
      "provenance_class": null
    },
    {
      "id": "G2i",
      "name": "Inference Cost per Solved Task at Fixed Reliability",
      "dimension": "G",
      "class": "contextual",
      "value": null,
      "unit": null,
      "display": null,
      "status": "not_measured",
      "direction": "flat",
      "polarity": "context",
      "uncertainty": null,
      "series": [],
      "method": "Not computed. No verified source in this run supplies the required input.",
      "source_id": null,
      "verification_class": "missing",
      "observed_at": null,
      "note": "Needs dated inference prices per model. Epoch's hardware file carries hardware release prices, not serving prices, and the price feed that would work is licence-constrained.",
      "provenance_class": null
    },
    {
      "id": "G3i",
      "name": "Frontier Training Compute",
      "dimension": "G",
      "class": "contextual",
      "value": 5.0000000000001e+26,
      "unit": "FLOP",
      "display": "5.0e+26 FLOP",
      "status": "measured",
      "direction": "down",
      "polarity": "context",
      "uncertainty": null,
      "series": [
        {
          "date": "1950-12-31",
          "value": 40.0
        },
        {
          "date": "1957-12-31",
          "value": 694894.9377361819
        },
        {
          "date": "1959-12-31",
          "value": 600000000.0
        },
        {
          "date": "1960-12-31",
          "value": 720000000.0
        },
        {
          "date": "1962-12-31",
          "value": 1559250.0
        },
        {
          "date": "1963-12-31",
          "value": 22500000.0
        },
        {
          "date": "1965-12-31",
          "value": 1080000.0
        },
        {
          "date": "1966-12-31",
          "value": 105917060.0
        },
        {
          "date": "1975-12-31",
          "value": 5184000.0
        },
        {
          "date": "1980-12-31",
          "value": 273738240.0
        },
        {
          "date": "1983-12-31",
          "value": 324000000.0
        },
        {
          "date": "1986-12-31",
          "value": 673920000.0
        },
        {
          "date": "1987-12-31",
          "value": 28328002560.0
        },
        {
          "date": "1988-12-31",
          "value": 296425000.0
        },
        {
          "date": "1989-12-31",
          "value": 1496338054440.0
        },
        {
          "date": "1990-12-31",
          "value": 78775200000.0
        },
        {
          "date": "1991-12-31",
          "value": 75474000000.0
        },
        {
          "date": "1992-12-31",
          "value": 18232157622832.703
        },
        {
          "date": "1993-12-31",
          "value": 12869570138112.0
        },
        {
          "date": "1994-12-31",
          "value": 18621900000000.0
        },
        {
          "date": "1995-12-31",
          "value": 195544800000.0
        },
        {
          "date": "1996-12-31",
          "value": 881733600000.0
        },
        {
          "date": "1997-12-31",
          "value": 31512000000000.0
        },
        {
          "date": "1998-12-31",
          "value": 2810937600000.0
        },
        {
          "date": "1999-12-31",
          "value": 8013600000000.0
        },
        {
          "date": "2000-12-31",
          "value": 6339000000000000.0
        },
        {
          "date": "2001-12-31",
          "value": 63000000000000.0
        },
        {
          "date": "2003-12-31",
          "value": 1666869200000000.0
        },
        {
          "date": "2004-12-31",
          "value": 2782080000000000.0
        },
        {
          "date": "2005-12-31",
          "value": 115848000000000.0
        },
        {
          "date": "2006-12-31",
          "value": 745200000000000.0
        },
        {
          "date": "2007-12-31",
          "value": 1.4494464e+18
        },
        {
          "date": "2008-12-31",
          "value": 1614600000.0
        },
        {
          "date": "2009-12-31",
          "value": 4181852160000000.0
        },
        {
          "date": "2010-12-31",
          "value": 5.396e+16
        },
        {
          "date": "2011-12-31",
          "value": 5.2e+16
        },
        {
          "date": "2012-12-31",
          "value": 6e+17
        },
        {
          "date": "2013-12-31",
          "value": 2.612736e+18
        },
        {
          "date": "2014-12-31",
          "value": 2.97600000001e+20
        },
        {
          "date": "2015-12-31",
          "value": 3.8e+20
        },
        {
          "date": "2016-12-31",
          "value": 6.620000000001e+21
        },
        {
          "date": "2017-12-31",
          "value": 8.43e+20
        },
        {
          "date": "2018-12-31",
          "value": 8.74395e+21
        },
        {
          "date": "2019-12-31",
          "value": 1.0773400001e+23
        },
        {
          "date": "2020-12-31",
          "value": 3.14e+23
        },
        {
          "date": "2021-12-31",
          "value": 2.047e+24
        },
        {
          "date": "2022-12-31",
          "value": 2.7415e+24
        },
        {
          "date": "2023-12-31",
          "value": 5.0000000001e+25
        },
        {
          "date": "2024-12-31",
          "value": 3.8e+25
        },
        {
          "date": "2025-12-31",
          "value": 5.0000000000001e+26
        },
        {
          "date": "2026-12-31",
          "value": 3.87e+25
        }
      ],
      "method": "The maximum estimated training compute in Epoch's Notable AI Models file, per publication year; 531 of 1041 rows carry both a compute estimate and a publication date. The reported value is the all-time maximum, reached by Grok 4 on 2025-07-09.",
      "source_id": "S2",
      "verification_class": "estimated",
      "observed_at": "2026-07-24",
      "note": "Context, never progress. These are Epoch estimates derived from parameter and token counts, not laboratory disclosures, and they are revisable without notice. The most recent year in the series (2026) is incomplete because compute estimates lag model releases, so a fall in the final bar is a coverage artefact and not a slowdown.",
      "provenance_class": "restated",
      "provenance_note": "Restates a figure the publisher reports itself."
    },
    {
      "id": "G4i",
      "name": "Score-vs-Inference-Compute Elasticity",
      "dimension": "G",
      "class": "contextual",
      "value": null,
      "unit": null,
      "display": null,
      "status": "not_measured",
      "direction": "flat",
      "polarity": "context",
      "uncertainty": null,
      "series": [],
      "method": "Not computed. No verified source in this run supplies the required input.",
      "source_id": null,
      "verification_class": "missing",
      "observed_at": null,
      "note": "Needs matched score and inference-compute pairs per model. The ECI file records an `optimized` flag but no compute or token budget per run.",
      "provenance_class": null
    },
    {
      "id": "H1",
      "name": "Open-Weight Lag",
      "dimension": "H",
      "class": "contextual",
      "value": null,
      "unit": null,
      "display": null,
      "status": "not_measured",
      "direction": "flat",
      "polarity": "context",
      "uncertainty": null,
      "series": [],
      "method": "Not computed. No verified source in this run supplies the required input.",
      "source_id": null,
      "verification_class": "missing",
      "observed_at": null,
      "note": "The ECI file carries no weights-access field, so the open versus closed split would have to be hand-maintained. No such join is maintained here.",
      "provenance_class": null
    },
    {
      "id": "H2",
      "name": "Frontier Concentration",
      "dimension": "H",
      "class": "contextual",
      "value": 0.3061,
      "unit": "HHI (0 to 1)",
      "display": "0.306 HHI",
      "status": "measured",
      "direction": "up",
      "polarity": "context",
      "uncertainty": null,
      "series": [
        {
          "date": "2024-05-31",
          "value": 0.5972
        },
        {
          "date": "2024-06-30",
          "value": 0.4595
        },
        {
          "date": "2024-07-31",
          "value": 0.3796
        },
        {
          "date": "2024-08-31",
          "value": 0.3486
        },
        {
          "date": "2024-09-30",
          "value": 0.3873
        },
        {
          "date": "2024-10-31",
          "value": 0.3932
        },
        {
          "date": "2024-11-30",
          "value": 0.3932
        },
        {
          "date": "2024-12-31",
          "value": 0.4896
        },
        {
          "date": "2025-01-31",
          "value": 0.5251
        },
        {
          "date": "2025-02-28",
          "value": 0.412
        },
        {
          "date": "2025-03-31",
          "value": 0.3297
        },
        {
          "date": "2025-04-30",
          "value": 0.3942
        },
        {
          "date": "2025-05-31",
          "value": 0.3417
        },
        {
          "date": "2025-06-30",
          "value": 0.3311
        },
        {
          "date": "2025-07-31",
          "value": 0.2656
        },
        {
          "date": "2025-08-31",
          "value": 0.3508
        },
        {
          "date": "2025-09-30",
          "value": 0.2731
        },
        {
          "date": "2025-10-31",
          "value": 0.289
        },
        {
          "date": "2025-11-30",
          "value": 0.2885
        },
        {
          "date": "2025-12-31",
          "value": 0.3035
        },
        {
          "date": "2026-01-31",
          "value": 0.3035
        },
        {
          "date": "2026-02-28",
          "value": 0.2981
        },
        {
          "date": "2026-03-31",
          "value": 0.3203
        },
        {
          "date": "2026-04-30",
          "value": 0.3194
        },
        {
          "date": "2026-05-31",
          "value": 0.327
        },
        {
          "date": "2026-06-30",
          "value": 0.2838
        },
        {
          "date": "2026-07-31",
          "value": 0.3061
        }
      ],
      "method": "Herfindahl-Hirschman index over developer shares of top-3 positions. For each benchmark in the A1 basket, the best score per model inside a trailing 365-day window is ranked and the top 3 models take one slot each; HHI is the sum of squared developer shares of all slots. Latest window: 83 slots held by 6 developers.",
      "source_id": "S4",
      "verification_class": "inferred",
      "observed_at": "2026-07-24",
      "note": "Developer labels are inferred from model-name prefixes by a fixed table in pipeline.py, not from a normalised upstream organisation identifier, so joint releases, subsidiaries and renames are not resolved. Labels-not-normalised. Shares are enumerated in notes.h2_shares.",
      "provenance_class": "derived",
      "provenance_note": "Computed by this tracker from published raw rows. No upstream source publishes this figure."
    },
    {
      "id": "I1",
      "name": "Capability-Controllability Divergence",
      "dimension": "I",
      "class": "tracked-separately",
      "value": null,
      "unit": null,
      "display": null,
      "status": "not_measured",
      "direction": "flat",
      "polarity": "context",
      "uncertainty": null,
      "series": [],
      "method": "Not computed. No verified source in this run supplies the required input.",
      "source_id": null,
      "verification_class": "missing",
      "observed_at": null,
      "note": "Needs capability and controllability scored on the same models under one protocol. The only source that did (Stanford HELM) is in maintenance mode and yields no fetchable content.",
      "provenance_class": null
    }
  ],
  "notes": {
    "row_counts": {
      "S4_eci_benchmarks_rows": 2059,
      "S4_distinct_benchmarks": 54,
      "S4_distinct_models": 213,
      "S5_metr_model_results": 26,
      "S2_notable_ai_models_rows": 1041,
      "S2_rows_with_compute_and_date": 531,
      "S3_ml_hardware_rows": 176
    },
    "date_range": {
      "as_of": "2026-07-24",
      "as_of_definition": "The newest observation date across all fetched sources. Every date-relative computation is anchored here, never on the wall clock, so a re-run on the same cache reproduces every value exactly.",
      "S4_first_observation": "2023-02-24",
      "S4_last_observation": "2026-07-24",
      "S5_first_model_release": "2019-02-14",
      "S5_last_model_release": "2026-04-07",
      "S2_last_publication_date": "2026-07-24"
    },
    "thresholds": [
      {
        "id": "T1",
        "dimension": "A",
        "indicator": "A1",
        "text": "A1 basket-median headroom at or below 0.50",
        "op": "<=",
        "tau": 0.5,
        "state": "MET",
        "observed": 0.3895
      },
      {
        "id": "T2",
        "dimension": "A",
        "indicator": "A1",
        "text": "A1 basket-median headroom at or below 0.25",
        "op": "<=",
        "tau": 0.25,
        "state": "UNMET",
        "observed": 0.3895
      },
      {
        "id": "T3",
        "dimension": "C",
        "indicator": "C2",
        "text": "C2 80% task time horizon at or above 240 minutes",
        "op": ">=",
        "tau": 240.0,
        "state": "UNMET",
        "observed": 185.9118
      },
      {
        "id": "T4",
        "dimension": "C",
        "indicator": "C1",
        "text": "C1 50% task time horizon at or above 1440 minutes",
        "op": ">=",
        "tau": 1440.0,
        "state": "UNMET",
        "observed": 1044.7801
      }
    ],
    "a1_basket": [
      {
        "benchmark": "FrontierMath-Tiers-1-3-v2-Private",
        "headroom": 0.1088
      },
      {
        "benchmark": "WeirdML",
        "headroom": 0.1124
      },
      {
        "benchmark": "Aider polyglot",
        "headroom": 0.12
      },
      {
        "benchmark": "GeoBench",
        "headroom": 0.12
      },
      {
        "benchmark": "FrontierMath-Tier-4-v2-Private",
        "headroom": 0.122
      },
      {
        "benchmark": "VPCT",
        "headroom": 0.135
      },
      {
        "benchmark": "Lech Mazur Writing",
        "headroom": 0.14
      },
      {
        "benchmark": "SWE-Bench verified",
        "headroom": 0.1653
      },
      {
        "benchmark": "SimpleBench",
        "headroom": 0.2172
      },
      {
        "benchmark": "SimpleQA Verified",
        "headroom": 0.227
      },
      {
        "benchmark": "GBAEval",
        "headroom": 0.2553
      },
      {
        "benchmark": "CursorBench",
        "headroom": 0.271
      },
      {
        "benchmark": "FrontierMath-2025-02-28-Private",
        "headroom": 0.3158
      },
      {
        "benchmark": "Chess Puzzles",
        "headroom": 0.36
      },
      {
        "benchmark": "Balrog",
        "headroom": 0.419
      },
      {
        "benchmark": "DeepResearch Bench",
        "headroom": 0.4469
      },
      {
        "benchmark": "FrontierCode",
        "headroom": 0.465
      },
      {
        "benchmark": "GDPval",
        "headroom": 0.503
      },
      {
        "benchmark": "APEX-Agents",
        "headroom": 0.504
      },
      {
        "benchmark": "GSO-Bench",
        "headroom": 0.5588
      },
      {
        "benchmark": "HLE",
        "headroom": 0.5626
      },
      {
        "benchmark": "The Agent Company",
        "headroom": 0.571
      },
      {
        "benchmark": "EBR-bench",
        "headroom": 0.6032
      },
      {
        "benchmark": "PostTrainBench",
        "headroom": 0.6571
      },
      {
        "benchmark": "CritPt",
        "headroom": 0.677
      },
      {
        "benchmark": "FrontierMath-Tier-4-2025-07-01-Private",
        "headroom": 0.6875
      },
      {
        "benchmark": "CL-bench",
        "headroom": 0.721
      },
      {
        "benchmark": "CL-bench Life",
        "headroom": 0.778
      }
    ],
    "a2_crossings": [
      {
        "benchmark": "GSM8K",
        "released": "2021-10-27",
        "crossed_on": "2023-03-15",
        "crossed_by": "GPT-4 (Mar 2023)",
        "score": 0.92,
        "months": 16.56
      },
      {
        "benchmark": "HellaSwag",
        "released": "2019-05-19",
        "crossed_on": "2023-03-15",
        "crossed_by": "GPT-4 (Mar 2023)",
        "score": 0.9373,
        "months": 45.87
      },
      {
        "benchmark": "ARC AI2",
        "released": "2018-03-14",
        "crossed_on": "2024-07-23",
        "crossed_by": "Llama 3.1-405B",
        "score": 0.9373,
        "months": 76.32
      },
      {
        "benchmark": "MATH level 5",
        "released": "2021-03-05",
        "crossed_on": "2024-12-05",
        "crossed_by": "o1",
        "score": 0.9471,
        "months": 45.04
      },
      {
        "benchmark": "Fiction.LiveBench",
        "released": "2025-02-21",
        "crossed_on": "2025-06-05",
        "crossed_by": "Gemini 2.5 Pro (Jun 2025)",
        "score": 0.917,
        "months": 3.42
      },
      {
        "benchmark": "OTIS Mock AIME 2024-2025",
        "released": "2024-12-19",
        "crossed_on": "2025-08-07",
        "crossed_by": "GPT-5",
        "score": 0.9138,
        "months": 7.59
      },
      {
        "benchmark": "GPQA diamond",
        "released": "2023-11-20",
        "crossed_on": "2025-11-18",
        "crossed_by": "Gemini 3 Pro",
        "score": 0.9015,
        "months": 23.95
      },
      {
        "benchmark": "ARC-AGI",
        "released": "2019-11-05",
        "crossed_on": "2025-12-11",
        "crossed_by": "GPT-5.2 Pro",
        "score": 0.905,
        "months": 73.2
      },
      {
        "benchmark": "Cybench",
        "released": "2024-08-15",
        "crossed_on": "2026-02-05",
        "crossed_by": "Claude Opus 4.6",
        "score": 0.93,
        "months": 17.71
      },
      {
        "benchmark": "Terminal Bench",
        "released": "2025-05-19",
        "crossed_on": "2026-04-16",
        "crossed_by": "Claude Opus 4.7",
        "score": 0.902,
        "months": 10.91
      }
    ],
    "a2_uncrossed": [
      {
        "benchmark": "ANLI",
        "released": "2019-10-31",
        "best_to_date": 0.3715
      },
      {
        "benchmark": "Aider polyglot",
        "released": "2024-12-21",
        "best_to_date": 0.88
      },
      {
        "benchmark": "BBH",
        "released": "2022-10-17",
        "best_to_date": 0.856
      },
      {
        "benchmark": "Balrog",
        "released": "2024-11-20",
        "best_to_date": 0.581
      },
      {
        "benchmark": "CadEval",
        "released": "2025-04-22",
        "best_to_date": 0.74
      },
      {
        "benchmark": "DeepResearch Bench",
        "released": "2025-06-13",
        "best_to_date": 0.5531
      },
      {
        "benchmark": "FrontierMath-2025-02-28-Private",
        "released": "2025-02-28",
        "best_to_date": 0.6842
      },
      {
        "benchmark": "GSO-Bench",
        "released": "2025-05-29",
        "best_to_date": 0.4412
      },
      {
        "benchmark": "GeoBench",
        "released": "2025-03-01",
        "best_to_date": 0.88
      },
      {
        "benchmark": "LAMBADA",
        "released": "2016-06-20",
        "best_to_date": 0.798
      },
      {
        "benchmark": "Lech Mazur Writing",
        "released": "2025-01-31",
        "best_to_date": 0.86
      },
      {
        "benchmark": "MMLU",
        "released": "2020-09-07",
        "best_to_date": 0.8413
      },
      {
        "benchmark": "OSWorld",
        "released": "2024-04-11",
        "best_to_date": 0.721
      },
      {
        "benchmark": "OpenBookQA",
        "released": "2018-09-08",
        "best_to_date": 0.84
      },
      {
        "benchmark": "PIQA",
        "released": "2019-11-26",
        "best_to_date": 0.774
      },
      {
        "benchmark": "SWE-Bench verified",
        "released": "2024-08-13",
        "best_to_date": 0.8347
      },
      {
        "benchmark": "ScienceQA",
        "released": "2022-09-20",
        "best_to_date": 0.8467
      },
      {
        "benchmark": "SimpleBench",
        "released": "2024-10-31",
        "best_to_date": 0.7828
      },
      {
        "benchmark": "The Agent Company",
        "released": "2024-12-18",
        "best_to_date": 0.429
      },
      {
        "benchmark": "TriviaQA",
        "released": "2017-05-09",
        "best_to_date": 0.876
      },
      {
        "benchmark": "VPCT",
        "released": "2025-01-30",
        "best_to_date": 0.865
      },
      {
        "benchmark": "WeirdML",
        "released": "2025-01-16",
        "best_to_date": 0.8876
      },
      {
        "benchmark": "Winogrande",
        "released": "2019-07-24",
        "best_to_date": 0.784
      }
    ],
    "a2_excluded_no_release_date": [
      "APEX-Agents",
      "ARC-AGI-2",
      "CL-bench",
      "CL-bench Life",
      "Chess Puzzles",
      "CritPt",
      "CursorBench",
      "EBR-bench",
      "ExploitBench",
      "FrontierCode",
      "FrontierMath-Tier-4-2025-07-01-Private",
      "FrontierMath-Tier-4-v2-Private",
      "FrontierMath-Tiers-1-3-v2-Private",
      "GBAEval",
      "GDPval",
      "HLE",
      "OSWorld 2.0",
      "PostTrainBench",
      "Remote Labor Index",
      "SimpleQA Verified",
      "Surface Evolver Bench"
    ],
    "d4_regressions": [
      {
        "developer": "Meta AI",
        "tier": "7b",
        "benchmark": "ARC AI2",
        "benchmark_release_date": "2018-03-14",
        "earlier_model": "LLaMA-7B",
        "earlier_date": "2023-02-24",
        "earlier_score": 0.3013,
        "later_model": "Llama 2-7B",
        "later_date": "2023-07-18",
        "later_score": 0.2787,
        "drop": 0.0227,
        "shared_benchmarks": 11
      },
      {
        "developer": "Meta AI",
        "tier": "7b",
        "benchmark": "PIQA",
        "benchmark_release_date": "2019-11-26",
        "earlier_model": "LLaMA-7B",
        "earlier_date": "2023-02-24",
        "earlier_score": 0.596,
        "later_model": "Llama 2-7B",
        "later_date": "2023-07-18",
        "later_score": 0.576,
        "drop": 0.02,
        "shared_benchmarks": 11
      },
      {
        "developer": "Microsoft",
        "tier": "flagship (no tier token)",
        "benchmark": "Winogrande",
        "benchmark_release_date": "2019-07-24",
        "earlier_model": "Phi-1.5",
        "earlier_date": "2023-09-11",
        "earlier_score": 0.468,
        "later_model": "Phi-2",
        "later_date": "2023-12-12",
        "later_score": 0.094,
        "drop": 0.374,
        "shared_benchmarks": 5
      },
      {
        "developer": "OpenAI",
        "tier": "mini",
        "benchmark": "Lech Mazur Writing",
        "benchmark_release_date": "2025-01-31",
        "earlier_model": "GPT-4o mini",
        "earlier_date": "2024-07-18",
        "earlier_score": 0.672,
        "later_model": "o1-mini",
        "later_date": "2024-09-12",
        "later_score": 0.649,
        "drop": 0.023,
        "shared_benchmarks": 8
      },
      {
        "developer": "OpenAI",
        "tier": "flagship (no tier token)",
        "benchmark": "Lech Mazur Writing",
        "benchmark_release_date": "2025-01-31",
        "earlier_model": "GPT-4o (Nov 2024)",
        "earlier_date": "2024-05-13",
        "earlier_score": 0.818,
        "later_model": "o1",
        "later_date": "2024-12-05",
        "later_score": 0.702,
        "drop": 0.116,
        "shared_benchmarks": 12
      },
      {
        "developer": "OpenAI",
        "tier": "flagship (no tier token)",
        "benchmark": "VPCT",
        "benchmark_release_date": "2025-01-30",
        "earlier_model": "GPT-4o (Nov 2024)",
        "earlier_date": "2024-05-13",
        "earlier_score": 0.1,
        "later_model": "o1",
        "later_date": "2024-12-05",
        "later_score": 0.055,
        "drop": 0.045,
        "shared_benchmarks": 12
      },
      {
        "developer": "Meta AI",
        "tier": "70b",
        "benchmark": "Balrog",
        "benchmark_release_date": "2024-11-20",
        "earlier_model": "Llama 3.1-70B",
        "earlier_date": "2024-07-23",
        "earlier_score": 0.279,
        "later_model": "Llama 3.3 70B",
        "later_date": "2024-12-06",
        "later_score": 0.23,
        "drop": 0.049,
        "shared_benchmarks": 6
      },
      {
        "developer": "OpenAI",
        "tier": "flagship (no tier token)",
        "benchmark": "GeoBench",
        "benchmark_release_date": "2025-03-01",
        "earlier_model": "o1",
        "earlier_date": "2024-12-05",
        "earlier_score": 0.8,
        "later_model": "o3",
        "later_date": "2024-12-20",
        "later_score": 0.74,
        "drop": 0.06,
        "shared_benchmarks": 15
      },
      {
        "developer": "OpenAI",
        "tier": "mini",
        "benchmark": "Lech Mazur Writing",
        "benchmark_release_date": "2025-01-31",
        "earlier_model": "o1-mini",
        "earlier_date": "2024-09-12",
        "earlier_score": 0.649,
        "later_model": "o3-mini",
        "later_date": "2025-01-31",
        "later_score": 0.617,
        "drop": 0.032,
        "shared_benchmarks": 10
      },
      {
        "developer": "Google DeepMind",
        "tier": "pro",
        "benchmark": "GPQA diamond",
        "benchmark_release_date": "2023-11-20",
        "earlier_model": "Gemini 2.5 Pro (Mar 2025)",
        "earlier_date": "2025-03-25",
        "earlier_score": 0.7845,
        "later_model": "Gemini 2.5 Pro (May 2025)",
        "later_date": "2025-05-06",
        "later_score": 0.5556,
        "drop": 0.229,
        "shared_benchmarks": 8
      },
      {
        "developer": "Google DeepMind",
        "tier": "pro",
        "benchmark": "VPCT",
        "benchmark_release_date": "2025-01-30",
        "earlier_model": "Gemini 2.5 Pro (Mar 2025)",
        "earlier_date": "2025-03-25",
        "earlier_score": 0.22,
        "later_model": "Gemini 2.5 Pro (May 2025)",
        "later_date": "2025-05-06",
        "later_score": 0.1075,
        "drop": 0.1125,
        "shared_benchmarks": 8
      },
      {
        "developer": "Google DeepMind",
        "tier": "flash",
        "benchmark": "OTIS Mock AIME 2024-2025",
        "benchmark_release_date": "2024-12-19",
        "earlier_model": "Gemini 2.5 Flash (Apr 2025)",
        "earlier_date": "2025-04-17",
        "earlier_score": 0.7303,
        "later_model": "Gemini 2.5 Flash (May 2025)",
        "later_date": "2025-05-20",
        "later_score": 0.708,
        "drop": 0.0222,
        "shared_benchmarks": 7
      },
      {
        "developer": "Anthropic",
        "tier": "sonnet",
        "benchmark": "Aider polyglot",
        "benchmark_release_date": "2024-12-21",
        "earlier_model": "Claude 3.7 Sonnet",
        "earlier_date": "2025-02-24",
        "earlier_score": 0.649,
        "later_model": "Claude Sonnet 4",
        "later_date": "2025-05-22",
        "later_score": 0.613,
        "drop": 0.036,
        "shared_benchmarks": 18
      },
      {
        "developer": "Anthropic",
        "tier": "sonnet",
        "benchmark": "Fiction.LiveBench",
        "benchmark_release_date": "2025-02-21",
        "earlier_model": "Claude 3.7 Sonnet",
        "earlier_date": "2025-02-24",
        "earlier_score": 0.833,
        "later_model": "Claude Sonnet 4",
        "later_date": "2025-05-22",
        "later_score": 0.469,
        "drop": 0.364,
        "shared_benchmarks": 18
      },
      {
        "developer": "Anthropic",
        "tier": "sonnet",
        "benchmark": "GeoBench",
        "benchmark_release_date": "2025-03-01",
        "earlier_model": "Claude 3.7 Sonnet",
        "earlier_date": "2025-02-24",
        "earlier_score": 0.68,
        "later_model": "Claude Sonnet 4",
        "later_date": "2025-05-22",
        "later_score": 0.37,
        "drop": 0.31,
        "shared_benchmarks": 18
      },
      {
        "developer": "Anthropic",
        "tier": "sonnet",
        "benchmark": "MATH level 5",
        "benchmark_release_date": "2021-03-05",
        "earlier_model": "Claude 3.7 Sonnet",
        "earlier_date": "2025-02-24",
        "earlier_score": 0.9116,
        "later_model": "Claude Sonnet 4",
        "later_date": "2025-05-22",
        "later_score": 0.8437,
        "drop": 0.068,
        "shared_benchmarks": 18
      },
      {
        "developer": "Anthropic",
        "tier": "sonnet",
        "benchmark": "VPCT",
        "benchmark_release_date": "2025-01-30",
        "earlier_model": "Claude 3.7 Sonnet",
        "earlier_date": "2025-02-24",
        "earlier_score": 0.085,
        "later_model": "Claude Sonnet 4",
        "later_date": "2025-05-22",
        "later_score": 0.01,
        "drop": 0.075,
        "shared_benchmarks": 18
      },
      {
        "developer": "Anthropic",
        "tier": "opus",
        "benchmark": "VPCT",
        "benchmark_release_date": "2025-01-30",
        "earlier_model": "Claude Opus 4",
        "earlier_date": "2025-05-22",
        "earlier_score": 0.07,
        "later_model": "Claude Opus 4.1",
        "later_date": "2025-08-05",
        "later_score": 0.025,
        "drop": 0.045,
        "shared_benchmarks": 10
      },
      {
        "developer": "OpenAI",
        "tier": "mini",
        "benchmark": "ARC-AGI",
        "benchmark_release_date": "2019-11-05",
        "earlier_model": "o4-mini",
        "earlier_date": "2025-04-16",
        "earlier_score": 0.587,
        "later_model": "GPT-5 mini",
        "later_date": "2025-08-07",
        "later_score": 0.5433,
        "drop": 0.0437,
        "shared_benchmarks": 13
      },
      {
        "developer": "OpenAI",
        "tier": "mini",
        "benchmark": "Fiction.LiveBench",
        "benchmark_release_date": "2025-02-21",
        "earlier_model": "o4-mini",
        "earlier_date": "2025-04-16",
        "earlier_score": 0.778,
        "later_model": "GPT-5 mini",
        "later_date": "2025-08-07",
        "later_score": 0.694,
        "drop": 0.084,
        "shared_benchmarks": 13
      },
      {
        "developer": "OpenAI",
        "tier": "mini",
        "benchmark": "GPQA diamond",
        "benchmark_release_date": "2023-11-20",
        "earlier_model": "o4-mini",
        "earlier_date": "2025-04-16",
        "earlier_score": 0.7281,
        "later_model": "GPT-5 mini",
        "later_date": "2025-08-07",
        "later_score": 0.6667,
        "drop": 0.0614,
        "shared_benchmarks": 13
      },
      {
        "developer": "OpenAI",
        "tier": "mini",
        "benchmark": "SimpleQA Verified",
        "benchmark_release_date": "not stated",
        "earlier_model": "o4-mini",
        "earlier_date": "2025-04-16",
        "earlier_score": 0.239,
        "later_model": "GPT-5 mini",
        "later_date": "2025-08-07",
        "later_score": 0.21,
        "drop": 0.029,
        "shared_benchmarks": 13
      },
      {
        "developer": "OpenAI",
        "tier": "mini",
        "benchmark": "VPCT",
        "benchmark_release_date": "2025-01-30",
        "earlier_model": "o4-mini",
        "earlier_date": "2025-04-16",
        "earlier_score": 0.3625,
        "later_model": "GPT-5 mini",
        "later_date": "2025-08-07",
        "later_score": 0.103,
        "drop": 0.2595,
        "shared_benchmarks": 13
      },
      {
        "developer": "OpenAI",
        "tier": "flagship (no tier token)",
        "benchmark": "CL-bench",
        "benchmark_release_date": "not stated",
        "earlier_model": "GPT-5.1",
        "earlier_date": "2025-11-13",
        "earlier_score": 0.237,
        "later_model": "GPT-5.2",
        "later_date": "2025-12-11",
        "later_score": 0.182,
        "drop": 0.055,
        "shared_benchmarks": 16
      },
      {
        "developer": "OpenAI",
        "tier": "flagship (no tier token)",
        "benchmark": "SimpleBench",
        "benchmark_release_date": "2024-10-31",
        "earlier_model": "GPT-5.1",
        "earlier_date": "2025-11-13",
        "earlier_score": 0.4384,
        "later_model": "GPT-5.2",
        "later_date": "2025-12-11",
        "later_score": 0.3496,
        "drop": 0.0888,
        "shared_benchmarks": 16
      },
      {
        "developer": "OpenAI",
        "tier": "pro",
        "benchmark": "SimpleBench",
        "benchmark_release_date": "2024-10-31",
        "earlier_model": "GPT-5 Pro",
        "earlier_date": "2025-10-07",
        "earlier_score": 0.5392,
        "later_model": "GPT-5.2 Pro",
        "later_date": "2025-12-11",
        "later_score": 0.4888,
        "drop": 0.0504,
        "shared_benchmarks": 5
      },
      {
        "developer": "OpenAI",
        "tier": "flagship (no tier token)",
        "benchmark": "SimpleQA Verified",
        "benchmark_release_date": "not stated",
        "earlier_model": "GPT-5.1",
        "earlier_date": "2025-11-13",
        "earlier_score": 0.489,
        "later_model": "GPT-5.2",
        "later_date": "2025-12-11",
        "later_score": 0.389,
        "drop": 0.1,
        "shared_benchmarks": 16
      },
      {
        "developer": "Google DeepMind",
        "tier": "flash",
        "benchmark": "ARC-AGI",
        "benchmark_release_date": "2019-11-05",
        "earlier_model": "Gemini 2.5 Flash (May 2025)",
        "earlier_date": "2025-05-20",
        "earlier_score": 0.333,
        "later_model": "Gemini 3 Flash",
        "later_date": "2025-12-17",
        "later_score": 0.215,
        "drop": 0.118,
        "shared_benchmarks": 5
      },
      {
        "developer": "Zhipu AI",
        "tier": "flagship (no tier token)",
        "benchmark": "OTIS Mock AIME 2024-2025",
        "benchmark_release_date": "2024-12-19",
        "earlier_model": "GLM-4.7",
        "earlier_date": "2025-12-22",
        "earlier_score": 0.8332,
        "later_model": "GLM-5",
        "later_date": "2026-02-11",
        "later_score": 0.7998,
        "drop": 0.0334,
        "shared_benchmarks": 10
      },
      {
        "developer": "xAI",
        "tier": "flagship (no tier token)",
        "benchmark": "Chess Puzzles",
        "benchmark_release_date": "not stated",
        "earlier_model": "Grok 4",
        "earlier_date": "2025-07-09",
        "earlier_score": 0.28,
        "later_model": "Grok 4.20",
        "later_date": "2026-02-17",
        "later_score": 0.24,
        "drop": 0.04,
        "shared_benchmarks": 8
      },
      {
        "developer": "xAI",
        "tier": "flagship (no tier token)",
        "benchmark": "SimpleQA Verified",
        "benchmark_release_date": "not stated",
        "earlier_model": "Grok 4",
        "earlier_date": "2025-07-09",
        "earlier_score": 0.479,
        "later_model": "Grok 4.20",
        "later_date": "2026-02-17",
        "later_score": 0.351,
        "drop": 0.128,
        "shared_benchmarks": 8
      },
      {
        "developer": "OpenAI",
        "tier": "flagship (no tier token)",
        "benchmark": "Chess Puzzles",
        "benchmark_release_date": "not stated",
        "earlier_model": "GPT-5.2",
        "earlier_date": "2025-12-11",
        "earlier_score": 0.49,
        "later_model": "GPT-5.4",
        "later_date": "2026-03-05",
        "later_score": 0.44,
        "drop": 0.05,
        "shared_benchmarks": 18
      },
      {
        "developer": "OpenAI",
        "tier": "flagship (no tier token)",
        "benchmark": "DeepResearch Bench",
        "benchmark_release_date": "2025-06-13",
        "earlier_model": "GPT-5.2",
        "earlier_date": "2025-12-11",
        "earlier_score": 0.4112,
        "later_model": "GPT-5.4",
        "later_date": "2026-03-05",
        "later_score": 0.3509,
        "drop": 0.0603,
        "shared_benchmarks": 18
      },
      {
        "developer": "OpenAI",
        "tier": "mini",
        "benchmark": "FrontierMath-Tier-4-v2-Private",
        "benchmark_release_date": "not stated",
        "earlier_model": "GPT-5 mini",
        "earlier_date": "2025-08-07",
        "earlier_score": 0.122,
        "later_model": "GPT-5.4 Mini",
        "later_date": "2026-03-17",
        "later_score": 0.0976,
        "drop": 0.0244,
        "shared_benchmarks": 8
      },
      {
        "developer": "Zhipu AI",
        "tier": "flagship (no tier token)",
        "benchmark": "GPQA diamond",
        "benchmark_release_date": "2023-11-20",
        "earlier_model": "GLM-5",
        "earlier_date": "2026-02-11",
        "earlier_score": 0.8376,
        "later_model": "GLM-5.1",
        "later_date": "2026-04-07",
        "later_score": 0.8064,
        "drop": 0.0312,
        "shared_benchmarks": 8
      },
      {
        "developer": "Anthropic",
        "tier": "opus",
        "benchmark": "SimpleBench",
        "benchmark_release_date": "2024-10-31",
        "earlier_model": "Claude Opus 4.6",
        "earlier_date": "2026-02-05",
        "earlier_score": 0.6112,
        "later_model": "Claude Opus 4.7",
        "later_date": "2026-04-16",
        "later_score": 0.5548,
        "drop": 0.0564,
        "shared_benchmarks": 18
      },
      {
        "developer": "Anthropic",
        "tier": "opus",
        "benchmark": "ARC-AGI-2",
        "benchmark_release_date": "not stated",
        "earlier_model": "Claude Opus 4.7",
        "earlier_date": "2026-04-16",
        "earlier_score": 0.758,
        "later_model": "Claude Opus 4.8",
        "later_date": "2026-05-28",
        "later_score": 0.7208,
        "drop": 0.0372,
        "shared_benchmarks": 18
      },
      {
        "developer": "Anthropic",
        "tier": "opus",
        "benchmark": "SimpleQA Verified",
        "benchmark_release_date": "not stated",
        "earlier_model": "Claude Opus 4.7",
        "earlier_date": "2026-04-16",
        "earlier_score": 0.506,
        "later_model": "Claude Opus 4.8",
        "later_date": "2026-05-28",
        "later_score": 0.395,
        "drop": 0.111,
        "shared_benchmarks": 18
      },
      {
        "developer": "Zhipu AI",
        "tier": "flagship (no tier token)",
        "benchmark": "OTIS Mock AIME 2024-2025",
        "benchmark_release_date": "2024-12-19",
        "earlier_model": "GLM-5.1",
        "earlier_date": "2026-04-07",
        "earlier_score": 0.9221,
        "later_model": "GLM-5.2",
        "later_date": "2026-06-16",
        "later_score": 0.8638,
        "drop": 0.0584,
        "shared_benchmarks": 8
      },
      {
        "developer": "Anthropic",
        "tier": "sonnet",
        "benchmark": "SimpleQA Verified",
        "benchmark_release_date": "not stated",
        "earlier_model": "Claude Sonnet 4.6",
        "earlier_date": "2026-02-17",
        "earlier_score": 0.29,
        "later_model": "Claude Sonnet 5",
        "later_date": "2026-06-30",
        "later_score": 0.25,
        "drop": 0.04,
        "shared_benchmarks": 7
      }
    ],
    "d4_diagnostics": {
      "pairs_evaluated": 70,
      "pairs_rejected_not_successor": 9,
      "comparable_cells": 611,
      "models_unmapped_to_developer": 17,
      "developer_tier_groups": 88
    },
    "d4_unmapped_models": [
      "Cerebras-GPT-13B",
      "Dolly 2.0-12b",
      "INTELLECT-1",
      "MPT-30B",
      "MPT-7B",
      "Muse Spark",
      "RedPajama-INCITE-7B-Base",
      "Stable Beluga 2",
      "StarCoder 2 15B",
      "StarCoder 2 3B",
      "StarCoder 2 7B",
      "XGen-7B",
      "internlm-20b",
      "internlm-7b",
      "open_llama_7b",
      "stablelm-tuned-alpha-7b",
      "vicuna-13b-v1.1"
    ],
    "h2_shares": {
      "slots": 83,
      "distinct_developers": 6,
      "shares": [
        {
          "developer": "OpenAI",
          "slots": 35,
          "share": 0.4217
        },
        {
          "developer": "Anthropic",
          "slots": 24,
          "share": 0.2892
        },
        {
          "developer": "Google DeepMind",
          "slots": 17,
          "share": 0.2048
        },
        {
          "developer": "DeepSeek",
          "slots": 3,
          "share": 0.0361
        },
        {
          "developer": "Zhipu AI",
          "slots": 3,
          "share": 0.0361
        },
        {
          "developer": "xAI",
          "slots": 1,
          "share": 0.012
        }
      ]
    },
    "measured_count": 8,
    "not_measured_count": 13,
    "caveats": [
      "Six of the eight measured indicators resolve to a single operator, Epoch AI. There is no non-Epoch feed of equivalent type for A1, A2, D4, G3i or H2, so operator failure is a single point of failure for most of this instrument.",
      "Epoch's ECI is a latent-trait fit that is re-anchored as benchmarks saturate, so historical values can move retroactively. The raw benchmark scores used here do not carry that property, but any comparison against a published ECI number does.",
      "The Epoch `date` column is the model date, not the date a score was measured, so every series here is dated by model rather than by evaluation.",
      "Scores in the ECI file come from a mix of Epoch's own evaluations, laboratory technical reports and third-party leaderboards. Indicators derived from that mix are classed `inferred` rather than independently-verified.",
      "Training compute figures are Epoch estimates derived from parameter and token counts, not laboratory disclosures, and carry wide error bars that are revisable without notice.",
      "METR's licence is not stated by the publisher. Attribution is given here as a matter of practice, not because a licence was found that requires it.",
      "D4 developer labels are inferred from model-name prefixes by a fixed table in pipeline.py. They are not normalised upstream identifiers, so joint releases, subsidiaries and renames are unresolved.",
      "The hardware file (S3) is fetched, hashed and registered but no indicator currently consumes it. It is retained as the named fallback input for G2i, which is not_measured.",
      "This is the first retained release, so the flag states REGRESSION (which needs D4's trailing four-period median) and INSTRUMENT-FAILURE (which needs a static position across releases) cannot be evaluated and are not raised."
    ]
  }
}
