[
  {
    "claim_id": "C01",
    "text": "SimpleBench contains more than 200 questions; its 83.7% human baseline came from nine participants.",
    "source_ids": [
      "S01"
    ],
    "status": "reported",
    "as_of": "2026-08-26",
    "confidence": "high"
  },
  {
    "claim_id": "C02",
    "text": "SWE-bench Verified retained 500 tasks after 93 professional developers reviewed 1,699 samples, with three annotations per sample.",
    "source_ids": [
      "S04"
    ],
    "status": "reported",
    "as_of": "2025-02-24",
    "confidence": "high"
  },
  {
    "claim_id": "C03",
    "text": "OpenAI's later audit found material issues in 59.4% of 138 audited SWE-bench Verified problems and recommended stopping frontier-launch reporting on the benchmark.",
    "source_ids": [
      "S05"
    ],
    "status": "reported",
    "as_of": "2026-08-26",
    "confidence": "high"
  },
  {
    "claim_id": "C04",
    "text": "Vals reported a 97.00% leading SWE-bench Verified result and seven of 86 models at 95% or higher on 19 August 2026.",
    "source_ids": [
      "S06"
    ],
    "status": "reported",
    "as_of": "2026-08-19",
    "confidence": "high"
  },
  {
    "claim_id": "C05",
    "text": "ARC-AGI-2 has 1,000 public training tasks and three 120-task evaluation sets: public, semi-private, and private.",
    "source_ids": [
      "S07"
    ],
    "status": "reported",
    "as_of": "2026-08-26",
    "confidence": "high"
  },
  {
    "claim_id": "C06",
    "text": "ARC-AGI-2 evaluation tasks were calibrated with more than 400 participants and each was solved pass@2 by at least two humans.",
    "source_ids": [
      "S07"
    ],
    "status": "reported",
    "as_of": "2026-08-26",
    "confidence": "high"
  },
  {
    "claim_id": "C07",
    "text": "Humanity's Last Exam moved from more than 70,000 submissions to 13,000 expert-review candidates, then 2,700 public questions and finally 2,500 after community and searchability review.",
    "source_ids": [
      "S10",
      "S11"
    ],
    "status": "reported",
    "as_of": "2025-04-03",
    "confidence": "high"
  },
  {
    "claim_id": "C08",
    "text": "GDPval contains 1,320 tasks across 44 occupations in nine sectors; task authors averaged more than 14 years of experience and tasks received five expert-review rounds on average.",
    "source_ids": [
      "S12"
    ],
    "status": "reported",
    "as_of": "2026-08-26",
    "confidence": "high"
  },
  {
    "claim_id": "C09",
    "text": "A 2026 study of 60 language-model benchmarks found nearly half saturated, with expert curation associated with greater resilience.",
    "source_ids": [
      "S14"
    ],
    "status": "reported",
    "as_of": "2026-08-06",
    "confidence": "high"
  },
  {
    "claim_id": "C10",
    "text": "Vals reported Claude Fable 5 at 93.18% on GPQA Diamond under its published accounting and 55.56% when refusals were counted as failures; the refusal rate was 41.92%.",
    "source_ids": [
      "S13"
    ],
    "status": "reported",
    "as_of": "2026-08-19",
    "confidence": "high"
  },
  {
    "claim_id": "C11",
    "text": "Arena's public flow presents two anonymous model responses and reveals their identities after the user votes.",
    "source_ids": [
      "S19"
    ],
    "status": "reported",
    "as_of": "2026-08-26",
    "confidence": "high"
  },
  {
    "claim_id": "C12",
    "text": "Arena reports confidence intervals and rank spreads; overlapping rank spreads indicate tied contenders under its method.",
    "source_ids": [
      "S20"
    ],
    "status": "reported",
    "as_of": "2026-02-28",
    "confidence": "high"
  },
  {
    "claim_id": "C13",
    "text": "Arena's 21 August 2026 Text snapshot contained 7,906,317 votes and 394 models; the first four models all had rank spreads containing rank one.",
    "source_ids": [
      "S21"
    ],
    "status": "reported",
    "as_of": "2026-08-21",
    "confidence": "high"
  },
  {
    "claim_id": "C14",
    "text": "Arena's freshness study analyzed 355,575 battles from May through December 2024 and found roughly 75% of daily prompts fresh at its 0.7 similarity threshold.",
    "source_ids": [
      "S23"
    ],
    "status": "reported",
    "as_of": "2025-08-02",
    "confidence": "high"
  },
  {
    "claim_id": "C15",
    "text": "Arena-Hard's first pipeline moved 200,000 user queries through more than 4,000 topics to 250 high-quality clusters and 500 prompts.",
    "source_ids": [
      "S22"
    ],
    "status": "reported",
    "as_of": "2024-04-02",
    "confidence": "high"
  },
  {
    "claim_id": "C16",
    "text": "In a recent seven-day Agent Arena slice, 160,480 tasks occurred across 128,244 sessions and 75.6% of sessions used at least one tool.",
    "source_ids": [
      "S26"
    ],
    "status": "reported",
    "as_of": "2026-06-04",
    "confidence": "high"
  },
  {
    "claim_id": "C17",
    "text": "LiveCodeBench continuously collects contest problems, LiveBench periodically adds questions from recent sources with objective grading, and Dynabench uses human-and-model-in-the-loop adversarial collection.",
    "source_ids": [
      "S30",
      "S31",
      "S32"
    ],
    "status": "reported",
    "as_of": "2026-08-26",
    "confidence": "high"
  },
  {
    "claim_id": "C18",
    "text": "Preference rankings can be affected by output style and sentiment, so they should not be read as pure capability measurements.",
    "source_ids": [
      "S24",
      "S25"
    ],
    "status": "reported",
    "as_of": "2026-08-26",
    "confidence": "medium"
  },
  {
    "claim_id": "C19",
    "text": "Artificial Analysis Intelligence Index v4.1.1 aggregates nine evaluations; its accessed 26 August 2026 page reported Claude Opus 5 with Adaptive Reasoning and Max Effort as the current published leader at 63.",
    "source_ids": [
      "S37",
      "S38"
    ],
    "status": "reported",
    "as_of": "2026-08-26",
    "confidence": "high"
  },
  {
    "claim_id": "C20",
    "text": "Artificial Analysis weights the v4.1.1 index across Agents 34%, Coding 24%, Scientific Reasoning 24%, and General 18%, and estimates an index-level 95% confidence interval below plus or minus one point.",
    "source_ids": [
      "S39"
    ],
    "status": "reported",
    "as_of": "2026-08-26",
    "confidence": "high"
  },
  {
    "claim_id": "C21",
    "text": "The Common Task Framework made progress comparable by giving competitors a shared task while independent evaluators scored performance against held-out data.",
    "source_ids": [
      "S40"
    ],
    "status": "reported",
    "as_of": "2015",
    "confidence": "high"
  },
  {
    "claim_id": "C22",
    "text": "The EU AI Act's framework for general-purpose AI models with systemic risk supplements compute thresholds with benchmarks and indicators for high-impact capability assessment.",
    "source_ids": [
      "S41"
    ],
    "status": "reported",
    "as_of": "2026-08-26",
    "confidence": "high"
  },
  {
    "claim_id": "C23",
    "text": "Anthropic activated ASL-3 protections for Claude Opus 4 as a precaution because it could not clearly rule out the relevant capability threshold, while continuing to evaluate the model.",
    "source_ids": [
      "S42"
    ],
    "status": "reported",
    "as_of": "2025-05-22",
    "confidence": "high"
  },
  {
    "claim_id": "C24",
    "text": "Modern benchmark practice grew from common-task evaluation: J. R. Pierce's review derailed United States machine translation funding for decades, IBM revived empirical translation work using bilingual Canadian parliamentary transcripts, and DARPA and NIST institutionalized shared tasks with sequestered evaluation data; the pattern later appeared in TREC and ImageNet.",
    "source_ids": [
      "S40",
      "S43",
      "S44",
      "S45"
    ],
    "status": "reported",
    "as_of": "2026-08-26",
    "confidence": "high"
  },
  {
    "claim_id": "C25",
    "text": "ARC-AGI-2 tasks are JSON records containing demonstration input-output pairs and held-out test inputs; grids use integer symbols zero through nine and range from 1 by 1 to 30 by 30.",
    "source_ids": [
      "S09"
    ],
    "status": "reported",
    "as_of": "2026-08-26",
    "confidence": "high"
  },
  {
    "claim_id": "C26",
    "text": "ARC-AGI-2 removed evaluation tasks susceptible to brute-force search and its public repository records later ambiguity and pixel-level task repairs in a dated changelog.",
    "source_ids": [
      "S07",
      "S09"
    ],
    "status": "reported",
    "as_of": "2026-08-26",
    "confidence": "high"
  },
  {
    "claim_id": "C27",
    "text": "ARC-AGI-2 allows two candidate outputs per test input and solves a task only when at least one candidate matches the expected dimensions and every cell for every test input.",
    "source_ids": [
      "S08",
      "S09"
    ],
    "status": "reported",
    "as_of": "2026-08-26",
    "confidence": "high"
  },
  {
    "claim_id": "C28",
    "text": "Arena's published comparison flow presents two model responses anonymously, records a user preference, and then reveals the model identities; the web-design and text battles shown are recorded battles from the Apache-2.0 arena-human-preference-55k dataset, while the code duel is an illustrative reconstruction.",
    "source_ids": [
      "S19",
      "S46"
    ],
    "status": "reported",
    "as_of": "2026-08-28",
    "confidence": "high"
  },
  {
    "claim_id": "C29",
    "text": "Across five widely quoted leaderboards accessed on 31 August 2026, the same five frontier models produce three different top-ranked models: Claude Opus 5 leads Vals' SWE-bench Verified and the Artificial Analysis Intelligence Index, Gemini 3.1 Pro Preview leads GPQA Diamond, and Claude Fable 5 leads LiveCodeBench v6 and the Text Arena overall board.",
    "source_ids": [
      "S48",
      "S49",
      "S50",
      "S51",
      "S52",
      "S53",
      "S54"
    ],
    "status": "reported",
    "as_of": "2026-08-31",
    "confidence": "high"
  },
  {
    "claim_id": "C30",
    "text": "GPT-5.6 Sol ranks second of 135 mirrored models on Vals' GPQA Diamond board at 95.20% and fiftieth of 140 mirrored models on Vals' LiveCodeBench v6 board at 82.60%, the largest cross-benchmark rank swing among the five models compared.",
    "source_ids": [
      "S53",
      "S54"
    ],
    "status": "reported",
    "as_of": "2026-08-31",
    "confidence": "medium"
  },
  {
    "claim_id": "C31",
    "text": "Leaderboard rows are model configurations rather than model families: the 27 August 2026 Text Arena board lists claude-opus-5-high at 1492 and claude-opus-5-max at 1488 as separate ranks, and the Artificial Analysis leaderboard lists Claude Opus 5 at 63, 63, 61, and 59 across its max, xhigh, high, and medium reasoning-effort rows.",
    "source_ids": [
      "S51",
      "S52"
    ],
    "status": "reported",
    "as_of": "2026-08-31",
    "confidence": "high"
  },
  {
    "claim_id": "C32",
    "text": "Vals AI reported GPT-5.6 Sol at 96.2% on the 500-task SWE-bench Verified split under its minimal bash-only agent harness and Claude Fable 5 at 93.18% on the 198-question GPQA Diamond split under the evaluator's published headline treatment.",
    "source_ids": [
      "S48",
      "S49"
    ],
    "status": "reported",
    "as_of": "2026-08-30",
    "confidence": "high"
  },
  {
    "claim_id": "C33",
    "text": "The official ARC Prize 2026 competition allows two predictions for each unseen output grid, awards one point when either exactly matches the hidden answer, and averages points across all task test outputs.",
    "source_ids": [
      "S55"
    ],
    "status": "reported",
    "as_of": "2026-09-01",
    "confidence": "high"
  },
  {
    "claim_id": "C34",
    "text": "ARC Prize reported Astra on ARC-AGI-3 Semi-Private at 99.9% with Provider Adapter and high reasoning, and 62.7% with Standard and max reasoning. At matched high reasoning Standard scored 54.8%. Provider Adapter preserves opaque reasoning state and compacts conversations; Standard uses visible notes.",
    "source_ids": [
      "S56"
    ],
    "status": "reported",
    "as_of": "2026-09-03",
    "confidence": "high"
  },
  {
    "claim_id": "C35",
    "text": "ARC-AGI-3 requires exploration, goal inference, modeling, and planning in unfamiliar interactive environments. It measures completion and action efficiency relative to a human baseline; public SK48 play is not a reproduction of the semi-private evaluation.",
    "source_ids": [
      "S56"
    ],
    "status": "reported",
    "as_of": "2026-09-03",
    "confidence": "high"
  },
  {
    "claim_id": "C36",
    "text": "ARC supports running its official game engine locally without per-move calls to the hosted game API. Local mode does not provide official online scorecards or shareable ARC replays.",
    "source_ids": [
      "S58"
    ],
    "status": "reported",
    "as_of": "2026-09-10",
    "confidence": "high"
  },
  {
    "claim_id": "C37",
    "text": "François Chollet introduced ARC in 2019 as grid tasks requiring rule inference from a few examples. ARC-AGI-2 followed in 2025 with new tasks including compositional reasoning. ARC-AGI-3 uses interactive environments where agents learn through actions and feedback.",
    "source_ids": [
      "S59",
      "S07",
      "S60"
    ],
    "status": "reported",
    "as_of": "2026-09-10",
    "confidence": "high"
  },
  {
    "claim_id": "C38",
    "text": "ARC-AGI-3 uses human-designed unfamiliar interactive environments to test exploration, rule and goal inference, planning, and efficient learning.",
    "source_ids": [
      "S64"
    ],
    "status": "reported",
    "as_of": "2026-09-16",
    "confidence": "high"
  },
  {
    "claim_id": "C39",
    "text": "ARC-AGI-3 was built in an in-house studio using concept review, implementation, internal and external testing. Later levels compose learned mechanics. Automated validation checks reliability, accidental solutions, and reproducible replay.",
    "source_ids": [
      "S61"
    ],
    "status": "reported",
    "as_of": "2026-09-16",
    "confidence": "high"
  },
  {
    "claim_id": "C40",
    "text": "The April 2026 human study involved 458 participants. Each retained environment was completed by at least two people; this does not claim every participant completed every game.",
    "source_ids": [
      "S63"
    ],
    "status": "reported",
    "as_of": "2026-09-16",
    "confidence": "high"
  },
  {
    "claim_id": "C41",
    "text": "The per-level baseline ranks completing first-time human players by action count and selects the upper median: the third entry among four or five finishers.",
    "source_ids": [
      "S62",
      "S63"
    ],
    "status": "reported",
    "as_of": "2026-09-16",
    "confidence": "high"
  },
  {
    "claim_id": "C42",
    "text": "The April 22 technical report lists 25 public demonstration, 55 semi-private, and 55 fully private ARC-AGI-3 environments. Protected games cover broader mechanics; public play is not the official evaluation.",
    "source_ids": [
      "S61"
    ],
    "status": "reported",
    "as_of": "2026-09-16",
    "confidence": "high"
  },
  {
    "claim_id": "C43",
    "text": "ARC-AGI-3 scores unfinished levels zero and completed levels min(1.15, (human actions / agent actions)^2). Level-index weights produce a game average capped by completed weights / all weights. The benchmark averages all game scores. Internal reasoning is not an environment action.",
    "source_ids": [
      "S62",
      "S61"
    ],
    "status": "reported",
    "as_of": "2026-09-16",
    "confidence": "high"
  },
  {
    "claim_id": "C44",
    "text": "Illustrative, not measured SK48 or model results: human action counts [12,16,20,24,32] yield baseline20; agent40 gives25%; agent10 is capped115%. Seven of eight levels cap at28/36=77.8%; game scores [1,28/36,.25] average67.6%.",
    "source_ids": [
      "S62"
    ],
    "status": "reported",
    "as_of": "2026-09-16",
    "confidence": "high"
  },
  {
    "claim_id": "C45",
    "text": "Selected results are transcribed from OpenAI’s official GPT-6 Astra release, checked 17 September 2026; the four-attempt SRE-Bench result comes from the accompanying narrative.",
    "source_ids": [
      "S65"
    ],
    "status": "reported",
    "as_of": "2026-09-17",
    "confidence": "high"
  },
  {
    "claim_id": "C46",
    "text": "The example percentages use distinct metrics: ARC3 action efficiency/completion, SRE task success, and an unwanted-behavior rate where lower is better.",
    "source_ids": [
      "S62",
      "S65"
    ],
    "status": "reported",
    "as_of": "2026-09-17",
    "confidence": "high"
  },
  {
    "claim_id": "C47",
    "text": "A reported run has settings including reasoning effort, tools and harness. The ARC3 launch footnote identifies a specific harness; ARC Prize reports results by harness and effort.",
    "source_ids": [
      "S65",
      "S56"
    ],
    "status": "reported",
    "as_of": "2026-09-16",
    "confidence": "high"
  },
  {
    "claim_id": "C48",
    "text": "OpenAI reports Astra SRE-Bench success of88.0% on one attempt and99.2% within four, a difference of11.2 percentage points.",
    "source_ids": [
      "S65"
    ],
    "status": "reported",
    "as_of": "2026-09-16",
    "confidence": "high"
  },
  {
    "claim_id": "C49",
    "text": "Missing cells in the official release are not zero scores; internal labels describe provenance rather than a numeric outcome.",
    "source_ids": [
      "S65"
    ],
    "status": "reported",
    "as_of": "2026-09-17",
    "confidence": "high"
  },
  {
    "claim_id": "C50",
    "text": "ECI summarizes performance on many benchmarks on a shared capability scale.",
    "source_ids": [
      "S66"
    ],
    "status": "reported",
    "as_of": "2026-09-16",
    "confidence": "high"
  },
  {
    "claim_id": "C51",
    "text": "ECI jointly estimates model capability and benchmark difficulty/slope using overlapping observations; the explanatory diagrams are schematic, not a reimplementation or numerical fit.",
    "source_ids": [
      "S67"
    ],
    "status": "reported",
    "as_of": "2026-09-16",
    "confidence": "high"
  },
  {
    "claim_id": "C52",
    "text": "The reader-supplied chart reports ECI169 for Astra. ECI is an abstract scale, not a percent or IQ score, and has no fixed maximum.",
    "source_ids": [
      "S68",
      "S70"
    ],
    "status": "reported",
    "as_of": "2026-09-16",
    "confidence": "high"
  },
  {
    "claim_id": "C53",
    "text": "The supplied chart reports Astra169 with90% bootstrap interval165–174. ECI documentation describes resampling observed benchmark evidence and refitting.",
    "source_ids": [
      "S67",
      "S70"
    ],
    "status": "reported",
    "as_of": "2026-09-16",
    "confidence": "high"
  },
  {
    "claim_id": "C54",
    "text": "The supplied historical chart reports a frontier trend of+15ECI/year and a90% prediction ribbon, distinct from the model point bootstrap interval.",
    "source_ids": [
      "S70"
    ],
    "status": "reported",
    "as_of": "2026-09-16",
    "confidence": "high"
  },
  {
    "claim_id": "C55",
    "text": "ECI compresses a performance profile, and specialized strengths may not be reflected by general standing. The two profile diagrams are illustrative and do not compute real equal ECI values.",
    "source_ids": [
      "S68"
    ],
    "status": "reported",
    "as_of": "2026-09-16",
    "confidence": "high"
  }
]
