{
  "schema_version": 2,
  "corrections": [
    {
      "correction_id": "gb2-20260713-match-total-001",
      "correction_type": "metric",
      "effective_at": "2026-07-18T17:28:10Z",
      "source_cut_generated_at": "2026-07-13T19:49:06Z",
      "field": "rated_match_count",
      "before": {
        "results_cut_id": "20260713_185718",
        "value": 56156,
        "canonical_results_sha256": "59e6912d25755d06647e56f4fc24462f4e90c7c27711d34c2b2e20209dbb2b4d"
      },
      "after": {
        "results_cut_id": "20260713_194906",
        "value": 56163,
        "canonical_results_sha256": "4e21a1a508679b79dfc1653d0143e43990564dc9f4f35bf3a142aba5bb9c87c7"
      },
      "reason": "Seven additional rated matches were admitted into the final launch-day results cut after the launch article's summary was prepared. Rankings and scoring policy were not silently rewritten; the article retains its launch snapshot and now labels it explicitly."
    },
    {
      "correction_id": "gb2-20260806-setting-identity-001",
      "correction_type": "release_coherence",
      "effective_at": "2026-08-07T12:33:34Z",
      "source_cut_generated_at": "2026-08-07T12:33:34Z",
      "field": "evaluated_setting_identity, ranking, and overall_eligibility",
      "before": {
        "results_cut_id": "20260806_094024",
        "canonical_results_sha256": "0c1621a7ffc5ee07402179ed197497f41591881aec3ca1950d1d58e41cb445cf"
      },
      "after": {
        "results_cut_id": "20260807_123334",
        "canonical_results_sha256": "57ebdef7ce4b39dc7ce37dd020b83370f8864dc2be9224daa5a02724a2e703f3"
      },
      "reason": "The corrected cut keeps deliberately heterogeneous provider reasoning settings while identifying rows by the complete retained provider request tuple. It recalculates published aggregates only where identity-bound selection, exact-request merging, or strict chronological supersession changes the contributing evidence. An opaque historical cohort is superseded only when the same effective request shape has a provably newer retained run; exact identities, tied timestamps, and incomplete chronology remain separate. Compatibility family counts change from 39/41/31 to 39/39/28 for Baseline/Balanced/Intensive. Reasoning Variants contains 111 current setting rows in both cuts, but the corrected identities and evidence differ; the intermediate identity-only cut contained 133 before provably older cohorts were removed. Overall remains 44 model families, and 108 Reasoning Variants matches without setting-identity provenance are excluded. The paired cut IDs and hashes preserve every row-level score and count for exact comparison.",
      "affected_public_release_ids": ["gb2-20260806-59b3356f5989"],
      "confirmed_findings": [
        "The affected release did not consistently preserve one canonical evaluated-setting identity across leaderboard rows, generation quality, costs, and per-game evidence.",
        "Compatibility views could expose more than one family row for a model in one bracket; the corrected cut merges only complete retained request tuples that are equal.",
        "Ranking and Overall eligibility were not represented by one shared authority across every public surface; corrected ranks are contiguous for eligible rows and null for provisional or partial rows.",
        "The corrected cut removes one unsupported Xiaomi medium setting, merges equivalent Claude Opus 4.5 historical thinking-budget aliases, and keeps genuinely distinct provider requests separate even when they share a broad reasoning bracket.",
        "Provably older opaque historical cohorts no longer appear in the current projection. This removes repeated Kimi K3 and Grok 4.5 rows while retaining tied-timestamp Gemini 3.6 Flash and LongCat 2.0 evidence that cannot safely be ordered."
      ],
      "unreproduced_allegations": [
        "A subsequent authored-route, sitemap, redirect, artifact-hash, and live read-only crawl did not reproduce a persistent cross-release canonical route. The original stale-route allegation is therefore not published as a confirmed fact."
      ],
      "preservation_statement": "The affected immutable release and its source cut remain preserved; the corrected release is a new immutable artifact and does not rewrite the prior release."
    },
    {
      "correction_id": "gb2-20260824-projection-coherence-001",
      "correction_type": "release_coherence",
      "effective_at": "2026-08-24T09:02:04Z",
      "source_cut_generated_at": "2026-08-24T08:54:07Z",
      "field": "latest cohort projection, scored reasoning variants, and public match telemetry",
      "before": {
        "results_cut_id": "20260823_205116",
        "canonical_results_sha256": "d780051938f41af2015f035dde8e070b9ea122e33931a2c2e28d0ad32a5696dc"
      },
      "after": {
        "results_cut_id": "20260824_085407",
        "canonical_results_sha256": "0af312135d2df50f5d3cab2f67c2a38c5f1d04b7f01b3391e331939c9c3d968e"
      },
      "reason": "The corrected cut applies exact historical run epochs when selecting the latest effective provider request, removes unevaluated Mixed aggregate placeholders, and exports match telemetry only from the row's own retained source identity. This eliminates repeated GPT-5.6 Sol, GPT-5.6 Luna, GPT-5.4, and Claude Sonnet 5 request rows, restores three scored Ox Alpha settings, and makes Overall totals explicitly unavailable where the underlying source does not support an exact attempted-match or opponent count. Qualification explanations now distinguish generation, fixed-coverage, and grouped-stability evidence instead of collapsing those states into a generation-quality claim.",
      "affected_public_release_ids": ["gb2-20260823-7d5648862dc8"],
      "confirmed_findings": [
        "Mapped direct4 run IDs lacked their audited generation epochs during latest-cohort selection, so older timestamped cohorts could remain visible beside the same effective request.",
        "One Claude Sonnet 5 correction retained an unknown embedded request identity even though its canonical correction columns and archived task proved the same none/toggle request as the newer direct4 cohort.",
        "Five unevaluated Mixed aggregate placeholders were published as zero-score rows: three Ox Alpha settings, Qwen 3.8 Max XHigh, and Inkling Small Max.",
        "Compatibility and Overall exports could mix row identities when deriving attempted matches, opponents, and fault totals, creating counts that did not describe the displayed aggregate.",
        "The corrected projection contains 48 Overall models and 121 evaluated reasoning variants, with no duplicate visible model/reasoning labels and no zero-entry Mixed rows."
      ],
      "unreproduced_allegations": [
        "The audit did not reproduce the apparent Xiaomi duplicate as an identity collision: its repeated broad brackets retain distinct effort/toggle request modes or distinct effective request values and remain separately labeled."
      ],
      "preservation_statement": "The affected immutable release and source cut remain preserved. The corrected cut and release are new immutable artifacts; no prior public result is rewritten."
    }
  ]
}
