[
  {
    "source_id": "S01",
    "publisher": "SimpleBench Team",
    "title": "SimpleBench",
    "canonical_url": "https://simple-bench.com/",
    "source_type": "project page",
    "publication_date": "not_reported",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary",
    "license": "not_reported",
    "archived_url": "not_reported",
    "notes": "Project description, settings, current model score, and nine-person human baseline."
  },
  {
    "source_id": "S02",
    "publisher": "Astropy",
    "title": "Issue #12906: Modeling compound model with multiple instances",
    "canonical_url": "https://github.com/astropy/astropy/issues/12906",
    "source_type": "issue",
    "publication_date": "2022-03-03",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary",
    "license": "BSD-3-Clause repository",
    "archived_url": "not_reported",
    "notes": "Public issue used in the SWE-bench interaction."
  },
  {
    "source_id": "S03",
    "publisher": "Astropy",
    "title": "Pull request #12907",
    "canonical_url": "https://github.com/astropy/astropy/pull/12907",
    "source_type": "pull request",
    "publication_date": "2022-03-03",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary",
    "license": "BSD-3-Clause repository",
    "archived_url": "not_reported",
    "notes": "Human fix and tests for issue #12906."
  },
  {
    "source_id": "S04",
    "publisher": "OpenAI",
    "title": "Introducing SWE-bench Verified",
    "canonical_url": "https://openai.com/index/introducing-swe-bench-verified/",
    "source_type": "methodology",
    "publication_date": "2024-08-13",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Construction, annotation, harness, and initial limitations."
  },
  {
    "source_id": "S05",
    "publisher": "OpenAI",
    "title": "Why SWE-bench Verified no longer measures frontier coding capabilities",
    "canonical_url": "https://openai.com/index/why-we-no-longer-evaluate-swe-bench-verified/",
    "source_type": "audit",
    "publication_date": "2026",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Audit of 138 inconsistent failures and contamination finding."
  },
  {
    "source_id": "S06",
    "publisher": "Vals AI",
    "title": "SWE-bench Verified",
    "canonical_url": "https://www.vals.ai/benchmarks/swebench",
    "source_type": "independent evaluator snapshot",
    "publication_date": "2026-08-19",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary evaluator",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Dated result and minimal bash-only harness."
  },
  {
    "source_id": "S07",
    "publisher": "ARC Prize Foundation",
    "title": "ARC-AGI-2",
    "canonical_url": "https://arcprize.org/arc-agi/2",
    "source_type": "benchmark documentation",
    "publication_date": "2025",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Dataset splits, calibration, human testing, and efficiency metric."
  },
  {
    "source_id": "S08",
    "publisher": "ARC Prize Foundation",
    "title": "ARC Prize testing policy",
    "canonical_url": "https://arcprize.org/policy",
    "source_type": "policy",
    "publication_date": "not_reported",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Official testing and private-set policy."
  },
  {
    "source_id": "S09",
    "publisher": "ARC Prize Foundation",
    "title": "ARC-AGI-2 repository",
    "canonical_url": "https://github.com/arcprize/ARC-AGI-2",
    "source_type": "dataset repository",
    "publication_date": "2025",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary",
    "license": "Apache-2.0",
    "archived_url": "not_reported",
    "notes": "Public task data and format, exact-match rule, license, task 0934a4d8 rendered in the interaction, and dated repair changelog."
  },
  {
    "source_id": "S10",
    "publisher": "Scale AI and Center for AI Safety",
    "title": "Humanity's Last Exam results",
    "canonical_url": "https://scale.com/blog/humanitys-last-exam-results",
    "source_type": "project report",
    "publication_date": "2025-01-24",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Initial 70,000 to 13,000 to 3,000 funnel and contributor figures."
  },
  {
    "source_id": "S11",
    "publisher": "Scale AI and Center for AI Safety",
    "title": "Humanity's Last Exam finalized leaderboard",
    "canonical_url": "https://labs.scale.com/leaderboard/humanitys_last_exam",
    "source_type": "methodology and leaderboard",
    "publication_date": "2025-04-03",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Final 2,500-item version after bug-bounty and searchability review."
  },
  {
    "source_id": "S12",
    "publisher": "OpenAI",
    "title": "GDPval",
    "canonical_url": "https://openai.com/index/gdpval/",
    "source_type": "methodology",
    "publication_date": "2025-09-25",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Task count, occupations, sectors, author experience, and review rounds."
  },
  {
    "source_id": "S13",
    "publisher": "Vals AI",
    "title": "GPQA Diamond",
    "canonical_url": "https://www.vals.ai/benchmarks/gpqa",
    "source_type": "independent evaluator snapshot",
    "publication_date": "2026-08-19",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary evaluator",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Published and refusal-corrected Claude Fable 5 results."
  },
  {
    "source_id": "S14",
    "publisher": "Akhtar et al.",
    "title": "When AI Benchmarks Plateau",
    "canonical_url": "https://arxiv.org/abs/2602.16763",
    "source_type": "peer-reviewed paper",
    "publication_date": "2026-08-06",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary research",
    "license": "CC BY 4.0",
    "archived_url": "not_reported",
    "notes": "Study of saturation across 60 language-model benchmarks."
  },
  {
    "source_id": "S15",
    "publisher": "Ott et al.",
    "title": "Mapping global dynamics of benchmark creation and saturation in artificial intelligence",
    "canonical_url": "https://arxiv.org/abs/2203.04592",
    "source_type": "peer-reviewed paper",
    "publication_date": "2022-10-07",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary research",
    "license": "arXiv license",
    "archived_url": "not_reported",
    "notes": "Historical map of 3,765 computer-vision and natural-language benchmarks."
  },
  {
    "source_id": "S16",
    "publisher": "Vals AI",
    "title": "MMLU Pro",
    "canonical_url": "https://www.vals.ai/benchmarks/mmlu_pro",
    "source_type": "independent evaluator snapshot",
    "publication_date": "2026-08-19",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary evaluator",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Current MMLU-Pro result and five-shot methodology."
  },
  {
    "source_id": "S17",
    "publisher": "Vals AI",
    "title": "LiveCodeBench",
    "canonical_url": "https://www.vals.ai/benchmarks/lcb",
    "source_type": "independent evaluator snapshot",
    "publication_date": "2026-08-19",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary evaluator",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Current LiveCodeBench result and version-six description."
  },
  {
    "source_id": "S18",
    "publisher": "Zheng et al.",
    "title": "Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference",
    "canonical_url": "https://arxiv.org/abs/2403.04132",
    "source_type": "peer-reviewed paper",
    "publication_date": "2024-03-07",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary research",
    "license": "arXiv license",
    "archived_url": "not_reported",
    "notes": "Original Arena method and static/live by ground-truth/preference taxonomy."
  },
  {
    "source_id": "S19",
    "publisher": "Arena",
    "title": "How Arena Works",
    "canonical_url": "https://arena.ai/how-it-works",
    "source_type": "product methodology",
    "publication_date": "2026",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Current prompt, anonymous comparison, vote, and reveal flow."
  },
  {
    "source_id": "S20",
    "publisher": "Arena",
    "title": "Arena's Ranking Method",
    "canonical_url": "https://arena.ai/blog/ranking-method",
    "source_type": "methodology",
    "publication_date": "2026-02-28",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Confidence intervals, raw rank, and rank spread."
  },
  {
    "source_id": "S21",
    "publisher": "Arena",
    "title": "Text Arena overall leaderboard",
    "canonical_url": "https://arena.ai/leaderboard/text/overall",
    "source_type": "live leaderboard snapshot",
    "publication_date": "2026-08-21",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Frozen V1 snapshot of votes, models, top scores, and rank spreads."
  },
  {
    "source_id": "S22",
    "publisher": "Arena",
    "title": "From Live Data to High-Quality Benchmarks: The Arena-Hard Pipeline",
    "canonical_url": "https://arena.ai/blog/arena-hard",
    "source_type": "methodology",
    "publication_date": "2024-04-02",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "200,000 queries to 4,000 topics to 250 clusters to 500 prompts."
  },
  {
    "source_id": "S23",
    "publisher": "Arena",
    "title": "How Many User Prompts are New?",
    "canonical_url": "https://arena.ai/blog/freshness",
    "source_type": "analysis",
    "publication_date": "2025-08-02",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "355,575-battle freshness analysis and threshold caveats."
  },
  {
    "source_id": "S24",
    "publisher": "Arena",
    "title": "Style Control",
    "canonical_url": "https://arena.ai/blog/style-control",
    "source_type": "analysis",
    "publication_date": "2024",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Analysis of style and length effects in preference data."
  },
  {
    "source_id": "S25",
    "publisher": "Arena",
    "title": "Sentiment Control",
    "canonical_url": "https://arena.ai/blog/sentiment-control/",
    "source_type": "analysis",
    "publication_date": "2025",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Observational analysis of sentiment and emoji effects."
  },
  {
    "source_id": "S26",
    "publisher": "Arena",
    "title": "Agent Arena: Causal Evaluation of Agents in the Real World",
    "canonical_url": "https://arena.ai/blog/agent-arena-methodology",
    "source_type": "methodology",
    "publication_date": "2026-06-04",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Real-session scale, agent components, outcome signals, and causal estimator."
  },
  {
    "source_id": "S27",
    "publisher": "Arena",
    "title": "Leaderboard Policy",
    "canonical_url": "https://arena.ai/blog/policy",
    "source_type": "policy",
    "publication_date": "2025",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Eligibility, provisional status, data release, and pre-release policy."
  },
  {
    "source_id": "S28",
    "publisher": "Arena",
    "title": "Arena Leaderboard Dataset",
    "canonical_url": "https://arena.ai/blog/arena-leaderboard-dataset",
    "source_type": "dataset documentation",
    "publication_date": "2026-04-02",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Historical leaderboard dataset covering multiple arenas and subsets."
  },
  {
    "source_id": "S29",
    "publisher": "Arena",
    "title": "Factuality in the Arena",
    "canonical_url": "https://arena.ai/blog/factuality-in-arena/",
    "source_type": "methodology update",
    "publication_date": "2026-07-14",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Example of a live evaluation system changing its target metric."
  },
  {
    "source_id": "S30",
    "publisher": "Jain et al.",
    "title": "LiveCodeBench",
    "canonical_url": "https://arxiv.org/abs/2403.07974",
    "source_type": "peer-reviewed paper",
    "publication_date": "2024-06-06",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary research",
    "license": "arXiv license",
    "archived_url": "not_reported",
    "notes": "Continuously collected contest problems and objective grading."
  },
  {
    "source_id": "S31",
    "publisher": "White et al.",
    "title": "LiveBench",
    "canonical_url": "https://arxiv.org/abs/2406.19314",
    "source_type": "peer-reviewed paper",
    "publication_date": "2025-04-18",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary research",
    "license": "arXiv license",
    "archived_url": "not_reported",
    "notes": "Frequently updated questions with objective ground truth."
  },
  {
    "source_id": "S32",
    "publisher": "Kiela et al.",
    "title": "Dynabench: Rethinking Benchmarking in NLP",
    "canonical_url": "https://arxiv.org/abs/2104.14337",
    "source_type": "peer-reviewed paper",
    "publication_date": "2021-04-28",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary research",
    "license": "arXiv license",
    "archived_url": "not_reported",
    "notes": "Human-and-model-in-the-loop adversarial data creation."
  },
  {
    "source_id": "S33",
    "publisher": "Chiang et al.",
    "title": "The Leaderboard Illusion",
    "canonical_url": "https://arxiv.org/abs/2504.20879",
    "source_type": "peer-reviewed paper",
    "publication_date": "2025-04-29",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary research",
    "license": "arXiv license",
    "archived_url": "not_reported",
    "notes": "Analysis of pre-release testing and differential access concerns."
  },
  {
    "source_id": "S34",
    "publisher": "Arena",
    "title": "Our response to The Leaderboard Illusion",
    "canonical_url": "https://arena.ai/blog/our-response/",
    "source_type": "response",
    "publication_date": "2025",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary response",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Arena's response and policy changes."
  },
  {
    "source_id": "S35",
    "publisher": "Raina et al.",
    "title": "Exploring and Mitigating Adversarial Attacks on LLM Leaderboards",
    "canonical_url": "https://arxiv.org/abs/2501.07493",
    "source_type": "peer-reviewed paper",
    "publication_date": "2025-01-13",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary research",
    "license": "arXiv license",
    "archived_url": "not_reported",
    "notes": "Simulated leaderboard manipulation and mitigation research."
  },
  {
    "source_id": "S36",
    "publisher": "SWE-bench",
    "title": "SWE-bench Verified dataset row astropy__astropy-12907",
    "canonical_url": "https://huggingface.co/datasets/princeton-nlp/SWE-bench_Verified",
    "source_type": "dataset",
    "publication_date": "2024",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary dataset",
    "license": "dataset and repository terms",
    "archived_url": "not_reported",
    "notes": "Problem statement, patch, and two fail-to-pass plus thirteen pass-to-pass test identifiers."
  },
  {
    "source_id": "S37",
    "publisher": "Artificial Analysis",
    "title": "Comparison of AI Models across Intelligence, Performance, and Price",
    "canonical_url": "https://artificialanalysis.ai/models/",
    "source_type": "independent evaluator snapshot",
    "publication_date": "not_reported",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary evaluator",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Current model table, Intelligence Index version, leader, score, and model configuration."
  },
  {
    "source_id": "S38",
    "publisher": "Artificial Analysis",
    "title": "Artificial Analysis Intelligence Index v4.1.1",
    "canonical_url": "https://artificialanalysis.ai/evaluations/artificial-analysis-intelligence-index",
    "source_type": "evaluation documentation",
    "publication_date": "not_reported",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary evaluator",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Composite scope, nine included evaluations, and current published leader."
  },
  {
    "source_id": "S39",
    "publisher": "Artificial Analysis",
    "title": "Artificial Analysis Intelligence Benchmarking Methodology",
    "canonical_url": "https://artificialanalysis.ai/methodology/intelligence-benchmarking",
    "source_type": "methodology",
    "publication_date": "not_reported",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary evaluator",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Category weights, component task counts and repeats, tool use, grading, and confidence-interval statement."
  },
  {
    "source_id": "S40",
    "publisher": "David Donoho",
    "title": "50 Years of Data Science",
    "canonical_url": "https://courses.csail.mit.edu/18.337/2015/docs/50YearsDataScience.pdf",
    "source_type": "research paper",
    "publication_date": "2015",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary research",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Historical account of the Common Task Framework and its role in empirical progress."
  },
  {
    "source_id": "S41",
    "publisher": "European Union",
    "title": "Regulation (EU) 2024/1689 — Artificial Intelligence Act",
    "canonical_url": "https://eur-lex.europa.eu/eli/reg/2024/1689/oj/eng",
    "source_type": "law",
    "publication_date": "2024-07-12",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary legal text",
    "license": "EU reuse terms",
    "archived_url": "not_reported",
    "notes": "Official text connecting systemic-risk classification and high-impact capability assessment to technical tools, benchmarks, and indicators."
  },
  {
    "source_id": "S42",
    "publisher": "Anthropic",
    "title": "Activating AI Safety Level 3 Protections",
    "canonical_url": "https://www.anthropic.com/news/activating-asl3-protections",
    "source_type": "deployment and safety report",
    "publication_date": "2025-05-22",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Describes the precautionary activation of ASL-3 protections for Claude Opus 4 when the relevant capability threshold could not be clearly ruled out."
  },
  {
    "source_id": "S43",
    "publisher": "National Institute of Standards and Technology",
    "title": "About the Text REtrieval Conference",
    "canonical_url": "https://trec.nist.gov/about.html",
    "source_type": "official program history",
    "publication_date": "not_reported",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary",
    "license": "United States government work",
    "archived_url": "not_reported",
    "notes": "Official account of TREC's 1992 launch, shared test collections, participant submissions, and NIST scoring."
  },
  {
    "source_id": "S44",
    "publisher": "Deng et al.",
    "title": "ImageNet: A Large-Scale Hierarchical Image Database",
    "canonical_url": "https://www.image-net.org/static_files/papers/imagenet_cvpr09.pdf",
    "source_type": "peer-reviewed paper",
    "publication_date": "2009",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary research",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "The original 2009 ImageNet paper describing the dataset and its intended use for visual recognition research and benchmarking."
  },
  {
    "source_id": "S45",
    "publisher": "ImageNet",
    "title": "ImageNet Large Scale Visual Recognition Challenge 2012",
    "canonical_url": "https://www.image-net.org/challenges/LSVRC/2012/",
    "source_type": "official challenge record",
    "publication_date": "2012",
    "accessed_date": "2026-08-26",
    "evidence_tier": "primary",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Official 2012 challenge record, release schedule, evaluation server, and competition results."
  },
  {
    "source_id": "S46",
    "publisher": "LMArena",
    "title": "arena-human-preference-55k dataset",
    "canonical_url": "https://huggingface.co/datasets/lmarena-ai/arena-human-preference-55k",
    "source_type": "dataset repository",
    "publication_date": "2024",
    "accessed_date": "2026-08-28",
    "evidence_tier": "primary",
    "license": "Apache-2.0",
    "archived_url": "not_reported",
    "notes": "Public dataset of recorded Arena battles: prompt, both anonymous model responses, model identities, and the recorded human vote. Rows 528442462 (web design) and 166595865 (text) are rendered in the interaction verbatim."
  },
  {
    "source_id": "S47",
    "publisher": "ARC Prize Foundation",
    "title": "ARC Prize leaderboard",
    "canonical_url": "https://arcprize.org/leaderboard",
    "source_type": "evaluator leaderboard",
    "publication_date": "2026",
    "accessed_date": "2026-08-28",
    "evidence_tier": "primary",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Official ARC-AGI leaderboard; the ARC-AGI-2 column and cost-per-task figures excerpted in data/arc-leaderboard.json, including the Human Panel row. No per-task model results are published."
  },
  {
    "source_id": "S48",
    "publisher": "Vals AI",
    "title": "SWE-bench Verified leaderboard",
    "canonical_url": "https://www.vals.ai/benchmarks/swebench",
    "source_type": "independent evaluator snapshot",
    "publication_date": "2026-08-30",
    "accessed_date": "2026-08-31",
    "evidence_tier": "primary evaluator",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Second dated access, later than S06. Board updated 30 August 2026; minimal bash-only agent harness over the 500-task split. Only a frontier subset of the 86 evaluated models was exposed on access."
  },
  {
    "source_id": "S49",
    "publisher": "Vals AI",
    "title": "GPQA Diamond leaderboard",
    "canonical_url": "https://www.vals.ai/benchmarks/gpqa",
    "source_type": "independent evaluator snapshot",
    "publication_date": "2026-08-30",
    "accessed_date": "2026-08-31",
    "evidence_tier": "primary evaluator",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Second dated access, later than S13. Board updated 30 August 2026; 24 of 136 models at 90% or higher; restates the Claude Fable 5 refusal-accounting caveat."
  },
  {
    "source_id": "S50",
    "publisher": "Vals AI",
    "title": "LiveCodeBench leaderboard",
    "canonical_url": "https://www.vals.ai/benchmarks/lcb",
    "source_type": "independent evaluator snapshot",
    "publication_date": "2026-08-30",
    "accessed_date": "2026-08-31",
    "evidence_tier": "primary evaluator",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Second dated access, later than S17. Version six; the leading models cluster within about two points."
  },
  {
    "source_id": "S51",
    "publisher": "Arena",
    "title": "Text Arena overall leaderboard, 27 August 2026",
    "canonical_url": "https://arena.ai/leaderboard/text",
    "source_type": "live leaderboard snapshot",
    "publication_date": "2026-08-27",
    "accessed_date": "2026-08-31",
    "evidence_tier": "primary",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Later rolling snapshot than S21, with 7,922,078 total votes. Leaderboard rows are served model configurations, not model families."
  },
  {
    "source_id": "S52",
    "publisher": "Artificial Analysis",
    "title": "Model leaderboard ranked by Intelligence Index",
    "canonical_url": "https://artificialanalysis.ai/leaderboards/models",
    "source_type": "independent evaluator snapshot",
    "publication_date": "not_reported",
    "accessed_date": "2026-08-31",
    "evidence_tier": "primary evaluator",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Ranked Intelligence Index table listing a separate row per reasoning-effort configuration. The index version was not restated on the accessed leaderboard page."
  },
  {
    "source_id": "S53",
    "publisher": "BenchLM",
    "title": "Vals GPQA Diamond mirror leaderboard",
    "canonical_url": "https://benchlm.ai/benchmarks/valsgpqadiamond",
    "source_type": "secondary leaderboard mirror",
    "publication_date": "2026-08-26",
    "accessed_date": "2026-08-31",
    "evidence_tier": "secondary mirror",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Per-model rows mirroring the Vals GPQA Diamond board for a 26 August 2026 snapshot of 135 models. Used because the Vals page did not expose per-model rows on access. Display-only on BenchLM and excluded from its own rankings."
  },
  {
    "source_id": "S54",
    "publisher": "BenchLM",
    "title": "Vals LiveCodeBench mirror leaderboard",
    "canonical_url": "https://benchlm.ai/benchmarks/valslivecodebench",
    "source_type": "secondary leaderboard mirror",
    "publication_date": "2026-08-19",
    "accessed_date": "2026-08-31",
    "evidence_tier": "secondary mirror",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Per-model rows mirroring the Vals LiveCodeBench board for a 19 August 2026 snapshot of 140 models. Display-only on BenchLM and excluded from its own rankings."
  },
  {
    "source_id": "S55",
    "publisher": "ARC Prize and Kaggle",
    "title": "ARC Prize 2026 — ARC-AGI-2 evaluation metric",
    "canonical_url": "https://www.kaggle.com/competitions/arc-prize-2026-arc-agi-2/",
    "source_type": "official competition metric",
    "publication_date": "2026",
    "accessed_date": "2026-09-01",
    "evidence_tier": "primary",
    "license": "competition terms",
    "archived_url": "not_reported",
    "notes": "Official per-output scoring rule: two predictions for each task test output, exact match earns one point, and the final score averages across all task test outputs."
  },
  {
    "source_id": "S56",
    "publisher": "ARC Prize",
    "title": "OpenAI’s GPT-6 Astra on ARC-AGI-3",
    "canonical_url": "https://arcprize.org/blog/astra",
    "source_type": "evaluation report",
    "publication_date": "2026-09-03",
    "accessed_date": "2026-09-10",
    "evidence_tier": "primary",
    "license": "not_reported",
    "archived_url": "not_reported",
    "notes": "Harness results by reasoning effort; action efficiency and human baseline. Public SK48 is not the semi-private evaluation."
  },
  {
    "source_id": "S57",
    "publisher": "ARC Prize",
    "title": "ARC-AGI-3 game documentation",
    "canonical_url": "https://docs.arcprize.org/games",
    "source_type": "documentation",
    "publication_date": "not_reported",
    "accessed_date": "2026-09-10",
    "evidence_tier": "primary",
    "license": "not_reported",
    "archived_url": "not_reported",
    "notes": "Public game API, action space, frames, and state. Used by the local SK48 proxy."
  },
  {
    "source_id": "S58",
    "publisher": "ARC Prize",
    "title": "ARC-AGI-3: Local vs Online",
    "canonical_url": "https://docs.arcprize.org/local-vs-online",
    "source_type": "documentation",
    "publication_date": "not_reported",
    "accessed_date": "2026-09-10",
    "evidence_tier": "primary",
    "license": "not_reported",
    "archived_url": "not_reported",
    "notes": "Official local execution mode avoids external game API calls; no online scorecards or shareable replays. The vendored SK48 source separately includes the MIT license and is pinned in vendor/arc3/manifest.json."
  },
  {
    "source_id": "S59",
    "publisher": "ARC Prize",
    "title": "ARC-AGI-1",
    "canonical_url": "https://arcprize.org/arc-agi/1",
    "source_type": "documentation",
    "publication_date": "not_reported",
    "accessed_date": "2026-09-10",
    "evidence_tier": "primary",
    "license": "not_reported",
    "archived_url": "not_reported",
    "notes": "Official overview supporting the evolution from static grid reasoning to interactive skill acquisition."
  },
  {
    "source_id": "S60",
    "publisher": "ARC Prize",
    "title": "ARC-AGI-3",
    "canonical_url": "https://arcprize.org/arc-agi/3",
    "source_type": "documentation",
    "publication_date": "not_reported",
    "accessed_date": "2026-09-10",
    "evidence_tier": "primary",
    "license": "not_reported",
    "archived_url": "not_reported",
    "notes": "Official overview supporting the evolution from static grid reasoning to interactive skill acquisition."
  },
  {
    "source_id": "S61",
    "publisher": "ARC Prize",
    "title": "ARC-AGI-3: A New Challenge for Frontier Agentic Intelligence",
    "canonical_url": "https://arcprize.org/media/ARC_AGI_3_Technical_Report.pdf",
    "source_type": "technical report",
    "publication_date": "2026-04-22",
    "accessed_date": "2026-09-16",
    "evidence_tier": "primary",
    "license": "not_reported",
    "archived_url": "not_reported",
    "notes": "Sections 3–5: studio, production, design, validation, dated splits, human calibration, and scoring."
  },
  {
    "source_id": "S62",
    "publisher": "ARC Prize",
    "title": "ARC-AGI-3 Scoring Methodology",
    "canonical_url": "https://docs.arcprize.org/methodology",
    "source_type": "documentation",
    "publication_date": "not_reported",
    "accessed_date": "2026-09-16",
    "evidence_tier": "primary",
    "license": "not_reported",
    "archived_url": "not_reported",
    "notes": "Current upper-median baseline, squared efficiency, 1.15 level cap, weighted game completion cap, and mean across games."
  },
  {
    "source_id": "S63",
    "publisher": "ARC Prize",
    "title": "Measuring Human Performance on ARC-AGI-3",
    "canonical_url": "https://arcprize.org/blog/arc-agi-3-human-dataset",
    "source_type": "research report",
    "publication_date": "2026-04-14",
    "accessed_date": "2026-09-16",
    "evidence_tier": "primary",
    "license": "not_reported",
    "archived_url": "not_reported",
    "notes": "458 participants; at least two solvers per retained environment; released public replays and scoring revision."
  },
  {
    "source_id": "S64",
    "publisher": "ARC Prize",
    "title": "Announcing ARC-AGI-3",
    "canonical_url": "https://arcprize.org/blog/arc-agi-3-launch",
    "source_type": "announcement",
    "publication_date": "2026-03-25",
    "accessed_date": "2026-09-16",
    "evidence_tier": "primary",
    "license": "not_reported",
    "archived_url": "not_reported",
    "notes": "Human-authored interactive environments and learning through exploration without supplied instructions."
  },
  {
    "source_id": "S65",
    "publisher": "OpenAI",
    "title": "GPT-6 Astra: A new generation of intelligence",
    "canonical_url": "https://openai.com/index/gpt-6-astra/",
    "source_type": "release report",
    "publication_date": "2026-09-03",
    "accessed_date": "2026-09-16",
    "evidence_tier": "primary",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Current release page: SRE single/four attempts, table metadata, grading direction, and evaluation footnotes."
  },
  {
    "source_id": "S66",
    "publisher": "Epoch AI",
    "title": "ECI documentation: overview",
    "canonical_url": "https://epoch.ai/data/eci-documentation",
    "source_type": "methodology",
    "publication_date": "not_reported",
    "accessed_date": "2026-09-16",
    "evidence_tier": "primary",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Composite index purpose and shared benchmark evidence."
  },
  {
    "source_id": "S67",
    "publisher": "Epoch AI",
    "title": "ECI documentation: methodology",
    "canonical_url": "https://epoch.ai/data/eci-documentation/methodology",
    "source_type": "methodology",
    "publication_date": "not_reported",
    "accessed_date": "2026-09-16",
    "evidence_tier": "primary",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Joint fit, scale calibration and bootstrap uncertainty; not an arithmetic mean of percentages."
  },
  {
    "source_id": "S68",
    "publisher": "Epoch AI",
    "title": "ECI documentation: FAQ",
    "canonical_url": "https://epoch.ai/data/eci-documentation/faq",
    "source_type": "methodology",
    "publication_date": "not_reported",
    "accessed_date": "2026-09-16",
    "evidence_tier": "primary",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Interpretation, no maximum, specialization, revisions as evidence changes."
  },
  {
    "source_id": "S69",
    "publisher": "Epoch AI",
    "title": "GPT-6 Astra model page",
    "canonical_url": "https://epoch.ai/models/gpt-6-astra",
    "source_type": "model record",
    "publication_date": "not_reported",
    "accessed_date": "2026-09-16",
    "evidence_tier": "primary",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "ECI 166 observed 2026-09-16; retained only as dated source context, not a live ranking."
  },
  {
    "source_id": "S70",
    "publisher": "Epoch AI",
    "title": "Launch-era Astra ECI chart: reader-supplied copy",
    "canonical_url": "https://epoch.ai/eci",
    "source_type": "reader-supplied chart",
    "publication_date": "not_reported",
    "accessed_date": "2026-09-16",
    "evidence_tier": "reader-supplied reproduction",
    "license": "CC BY (as marked on supplied chart)",
    "archived_url": "not_reported",
    "notes": "Supplied file HRUTmX-aUAAymEX.jpg; preserved at references/epoch-astra-launch.jpg. Chart says 169, 90% bootstrap CI165–174, frontier trend+15/year, 90% prediction band. Exact original post URL was not supplied; canonical URL is the publisher index, not a claim to locate this historical version."
  },
  {
    "source_id": "S71",
    "publisher": "OpenAI (as attributed in supplied image)",
    "title": "Launch-era Astra release table: reader-supplied copy",
    "canonical_url": "https://openai.com/index/gpt-6-astra/",
    "source_type": "reader-supplied screenshot",
    "publication_date": "not_reported",
    "accessed_date": "2026-09-16",
    "evidence_tier": "reader-supplied reproduction",
    "license": "publisher terms",
    "archived_url": "not_reported",
    "notes": "Supplied file HRT9KFUaYAAe3xu.jpg; preserved at references/astra-release-table.jpg. Snapshot differs from current release page and does not include full footnotes. Canonical URL links to current primary release report, not this exact historical table."
  }
]
