[
  {"result_id":"R01","benchmark_id":"swebench_verified","benchmark_version":"500-task verified split","task_split":"verified","model_id":"claude-opus-5","exact_model_version":"not_reported","provider":"Anthropic","evaluator":"Vals AI","run_date":"not_reported","published_date":"2026-08-19","as_of":"2026-08-19","prompt":"system prompt described by evaluator; exact text not_reported","system_prompt_status":"described, exact text not_reported","few_shot_configuration":"not_applicable","reasoning_effort":"provider default","temperature":"not_reported","max_tokens":"highest provider-supported value","tools":"bash only","agent_scaffold":"mini-swe-agent","harness_container":"isolated cloud sandbox and SWE-bench Docker tasks","score":97,"unit":"percent resolved","numerator":"not_reported","denominator":500,"confidence_interval":"not_reported","cost":"not_reported","latency":"reported by evaluator but not captured in V1","refusal_policy":"not_reported","retry_policy":"not_reported","failure_policy":"all required tests must pass","source_ids":["S06"],"fetched_at":"2026-08-26","coverage_state":"reported","notes":"A model-plus-scaffold result; OpenAI separately recommends retiring this benchmark for frontier launches."},
  {"result_id":"R02","benchmark_id":"gpqa_diamond","benchmark_version":"198-question Diamond split","task_split":"diamond","model_id":"claude-fable-5","exact_model_version":"not_reported","provider":"Anthropic","evaluator":"Vals AI","run_date":"not_reported","published_date":"2026-08-19","as_of":"2026-08-19","prompt":"zero- and few-shot methods described by evaluator","system_prompt_status":"not_reported","few_shot_configuration":"published headline treatment","reasoning_effort":"not_reported","temperature":"not_reported","max_tokens":"not_reported","tools":"none reported","agent_scaffold":"not_applicable","harness_container":"Vals evaluator","score":93.18,"unit":"percent correct","numerator":"not_reported","denominator":198,"confidence_interval":"not_reported","cost":"not_reported","latency":"not_reported","refusal_policy":"refusal-triggered fallbacks counted as successes","retry_policy":"not_reported","failure_policy":"publisher treatment","source_ids":["S13"],"fetched_at":"2026-08-26","coverage_state":"reported","notes":"Vals flags this headline treatment as misleading for this model."},
  {"result_id":"R03","benchmark_id":"gpqa_diamond","benchmark_version":"198-question Diamond split","task_split":"diamond","model_id":"claude-fable-5","exact_model_version":"not_reported","provider":"Anthropic","evaluator":"Vals AI","run_date":"not_reported","published_date":"2026-08-19","as_of":"2026-08-19","prompt":"same run as R02","system_prompt_status":"not_reported","few_shot_configuration":"same run as R02","reasoning_effort":"not_reported","temperature":"not_reported","max_tokens":"not_reported","tools":"none reported","agent_scaffold":"not_applicable","harness_container":"Vals evaluator","score":55.56,"unit":"percent correct","numerator":"not_reported","denominator":198,"confidence_interval":"not_reported","cost":"not_reported","latency":"not_reported","refusal_policy":"refusals counted as failures; 41.92% refusal rate","retry_policy":"not_reported","failure_policy":"corrected treatment presented by Vals","source_ids":["S13"],"fetched_at":"2026-08-26","coverage_state":"calculated_from_reported_values","notes":"Vals-provided corrected treatment, not an independent MTS recalculation."},
  {"result_id":"R04","benchmark_id":"gpqa_diamond","benchmark_version":"198-question Diamond split","task_split":"diamond","model_id":"gemini-3.1-pro-preview-02-26","exact_model_version":"Gemini 3.1 Pro Preview (02/26)","provider":"Google","evaluator":"Vals AI","run_date":"not_reported","published_date":"2026-08-19","as_of":"2026-08-19","prompt":"Vals GPQA methodology","system_prompt_status":"not_reported","few_shot_configuration":"zero- and five-shot methods described","reasoning_effort":"not_reported","temperature":"not_reported","max_tokens":"not_reported","tools":"none reported","agent_scaffold":"not_applicable","harness_container":"Vals evaluator","score":95.45,"unit":"percent correct","numerator":"not_reported","denominator":198,"confidence_interval":"not_reported","cost":"not_reported","latency":"not_reported","refusal_policy":"not_reported","retry_policy":"not_reported","failure_policy":"answer parsing described by evaluator","source_ids":["S13"],"fetched_at":"2026-08-26","coverage_state":"reported","notes":"Current Vals leader in the dated snapshot."},
  {"result_id":"R05","benchmark_id":"mmlu_pro","benchmark_version":"Vals implementation","task_split":"14-subject overall","model_id":"claude-opus-5","exact_model_version":"not_reported","provider":"Anthropic","evaluator":"Vals AI","run_date":"not_reported","published_date":"2026-08-19","as_of":"2026-08-19","prompt":"five-shot chain-of-thought","system_prompt_status":"example prompts published","few_shot_configuration":"5 examples per category","reasoning_effort":"not_reported","temperature":"not_reported","max_tokens":"not_reported","tools":"none reported","agent_scaffold":"not_applicable","harness_container":"Vals evaluator","score":91.59,"unit":"percent correct","numerator":"not_reported","denominator":"more than 12,000","confidence_interval":"not_reported","cost":"not_reported","latency":"not_reported","refusal_policy":"not_reported","retry_policy":"not_reported","failure_policy":"regex answer parsing","source_ids":["S16"],"fetched_at":"2026-08-26","coverage_state":"reported","notes":"Approximately 15 million input tokens per model reported by Vals."},
  {"result_id":"R06","benchmark_id":"livecodebench","benchmark_version":"v6","task_split":"overall","model_id":"claude-fable-5","exact_model_version":"not_reported","provider":"Anthropic","evaluator":"Vals AI","run_date":"not_reported","published_date":"2026-08-19","as_of":"2026-08-19","prompt":"not_reported","system_prompt_status":"not_reported","few_shot_configuration":"not_reported","reasoning_effort":"not_reported","temperature":"not_reported","max_tokens":"not_reported","tools":"none reported","agent_scaffold":"not_applicable","harness_container":"hidden functional tests","score":89.78,"unit":"percent correct","numerator":"not_reported","denominator":"over 1,000","confidence_interval":"not_reported","cost":"not_reported","latency":"not_reported","refusal_policy":"not_reported","retry_policy":"not_reported","failure_policy":"must pass hidden tests","source_ids":["S17"],"fetched_at":"2026-08-26","coverage_state":"reported","notes":"Easy split described as effectively solved; hard problems drive separation."},
  {"result_id":"R07","benchmark_id":"chatbot_arena","benchmark_version":"Text overall snapshot","task_split":"overall, no adjustment","model_id":"claude-fable-5","exact_model_version":"claude-fable-5","provider":"Anthropic","evaluator":"Arena","run_date":"rolling through 2026-08-21","published_date":"2026-08-21","as_of":"2026-08-21","prompt":"live user prompts","system_prompt_status":"varies by served model configuration","few_shot_configuration":"not_applicable","reasoning_effort":"model configuration","temperature":"not_reported","max_tokens":"not_reported","tools":"text chat","agent_scaffold":"Arena battle mode","harness_container":"Arena serving stack","score":1508,"unit":"Arena score","numerator":"24,331 model votes","denominator":"7,906,317 total leaderboard votes","confidence_interval":"±5","cost":"$10 input / $50 output per million tokens listed","latency":"not_reported","refusal_policy":"users may select both bad","retry_policy":"continuous battles","failure_policy":"pairwise preference outcome","source_ids":["S21"],"fetched_at":"2026-08-26","coverage_state":"reported","notes":"Raw rank 1, rank spread 1–4; not statistically unique at rank one."},
  {"result_id":"R08","benchmark_id":"aa_intelligence_index","benchmark_version":"v4.1.1","task_split":"nine-evaluation weighted composite","model_id":"claude-opus-5-adaptive-max","exact_model_version":"Claude Opus 5 (Adaptive Reasoning, Max Effort)","provider":"Anthropic","evaluator":"Artificial Analysis","run_date":"not_reported","published_date":"not_reported","as_of":"2026-08-26","prompt":"component prompts described by evaluator; exact suite prompt set partially reported","system_prompt_status":"component-specific; not fully reported","few_shot_configuration":"varies by component evaluation","reasoning_effort":"Adaptive Reasoning, Max Effort","temperature":"component-specific; not_reported in composite record","max_tokens":"component-specific; not_reported in composite record","tools":"varies by component; see evaluator methodology","agent_scaffold":"Artificial Analysis evaluation harness","harness_container":"not_reported","score":63,"unit":"0–100 weighted Intelligence Index","numerator":"not_applicable","denominator":"nine evaluations","confidence_interval":"index-level 95% confidence interval estimated below ±1 point","cost":"reported on evaluator comparison; not captured in frozen V1","latency":"reported on evaluator comparison; not captured in frozen V1","refusal_policy":"component-specific; not_reported in composite record","retry_policy":"component-specific repeats published in methodology","failure_policy":"component-specific scoring aggregated by published category weights","source_ids":["S37","S38","S39"],"fetched_at":"2026-08-26","coverage_state":"reported","notes":"Current published leader in the accessed Artificial Analysis Intelligence Index v4.1.1 snapshot; this composite score is not comparable to any Vals percentage."},
  {"result_id":"R09","benchmark_id":"swebench_verified","benchmark_version":"500-task verified split","task_split":"verified","model_id":"claude-opus-5","exact_model_version":"not_reported","provider":"Anthropic","evaluator":"Vals AI","run_date":"not_reported","published_date":"2026-08-30","as_of":"2026-08-30","prompt":"system prompt described by evaluator; exact text not_reported","system_prompt_status":"described, exact text not_reported","few_shot_configuration":"not_applicable","reasoning_effort":"not_reported","temperature":"not_reported","max_tokens":"not_reported","tools":"bash only","agent_scaffold":"minimal bash-tool-only agent harness","harness_container":"isolated SWE-bench Docker task containers","score":97,"unit":"percent resolved","numerator":"not_reported","denominator":500,"confidence_interval":"not_reported","cost":"not_reported","latency":"not_reported","refusal_policy":"not_reported","retry_policy":"not_reported","failure_policy":"all required tests must pass","source_ids":["S48"],"fetched_at":"2026-08-31","coverage_state":"reported","notes":"Board leader on access. Matches the 19 August 2026 figure in R01, so the number is stable across two dated accesses."},
  {"result_id":"R10","benchmark_id":"swebench_verified","benchmark_version":"500-task verified split","task_split":"verified","model_id":"kimi-k3","exact_model_version":"not_reported","provider":"Moonshot AI","evaluator":"Vals AI","run_date":"not_reported","published_date":"2026-08-30","as_of":"2026-08-30","prompt":"system prompt described by evaluator; exact text not_reported","system_prompt_status":"described, exact text not_reported","few_shot_configuration":"not_applicable","reasoning_effort":"not_reported","temperature":"not_reported","max_tokens":"not_reported","tools":"bash only","agent_scaffold":"minimal bash-tool-only agent harness","harness_container":"isolated SWE-bench Docker task containers","score":93.4,"unit":"percent resolved","numerator":"not_reported","denominator":500,"confidence_interval":"not_reported","cost":"not_reported","latency":"not_reported","refusal_policy":"not_reported","retry_policy":"not_reported","failure_policy":"all required tests must pass","source_ids":["S48"],"fetched_at":"2026-08-31","coverage_state":"reported","notes":"Third on the board behind DeepSeek V4 Pro 0813 at 96.40%."},
  {"result_id":"R11","benchmark_id":"gpqa_diamond","benchmark_version":"198-question Diamond split","task_split":"diamond","model_id":"gemini-3.1-pro-preview-02-26","exact_model_version":"not_reported","provider":"Google","evaluator":"Vals AI, read via BenchLM mirror","run_date":"not_reported","published_date":"2026-08-26","as_of":"2026-08-26","prompt":"zero- and few-shot chain-of-thought methods described by evaluator","system_prompt_status":"not_reported","few_shot_configuration":"published headline treatment","reasoning_effort":"not_reported","temperature":"not_reported","max_tokens":"not_reported","tools":"none reported","agent_scaffold":"not_applicable","harness_container":"not_reported","score":95.45,"unit":"percent correct","numerator":"not_reported","denominator":198,"confidence_interval":"not_reported","cost":"not_reported","latency":"not_reported","refusal_policy":"not_reported","retry_policy":"not_reported","failure_policy":"publisher treatment","source_ids":["S53","S49"],"fetched_at":"2026-08-31","coverage_state":"reported","notes":"Mirror leader, consistent with R04 under the earlier access."},
  {"result_id":"R12","benchmark_id":"gpqa_diamond","benchmark_version":"198-question Diamond split","task_split":"diamond","model_id":"gpt-5.6-sol","exact_model_version":"not_reported","provider":"OpenAI","evaluator":"Vals AI, read via BenchLM mirror","run_date":"not_reported","published_date":"2026-08-26","as_of":"2026-08-26","prompt":"zero- and few-shot chain-of-thought methods described by evaluator","system_prompt_status":"not_reported","few_shot_configuration":"published headline treatment","reasoning_effort":"not_reported","temperature":"not_reported","max_tokens":"not_reported","tools":"none reported","agent_scaffold":"not_applicable","harness_container":"not_reported","score":95.2,"unit":"percent correct","numerator":"not_reported","denominator":198,"confidence_interval":"not_reported","cost":"not_reported","latency":"not_reported","refusal_policy":"not_reported","retry_policy":"not_reported","failure_policy":"publisher treatment","source_ids":["S53","S49"],"fetched_at":"2026-08-31","coverage_state":"reported","notes":"Rank 2, within 0.25 points of the leader."},
  {"result_id":"R13","benchmark_id":"gpqa_diamond","benchmark_version":"198-question Diamond split","task_split":"diamond","model_id":"claude-opus-5","exact_model_version":"not_reported","provider":"Anthropic","evaluator":"Vals AI, read via BenchLM mirror","run_date":"not_reported","published_date":"2026-08-26","as_of":"2026-08-26","prompt":"zero- and few-shot chain-of-thought methods described by evaluator","system_prompt_status":"not_reported","few_shot_configuration":"published headline treatment","reasoning_effort":"not_reported","temperature":"not_reported","max_tokens":"not_reported","tools":"none reported","agent_scaffold":"not_applicable","harness_container":"not_reported","score":93.43,"unit":"percent correct","numerator":"not_reported","denominator":198,"confidence_interval":"not_reported","cost":"not_reported","latency":"not_reported","refusal_policy":"not_reported","retry_policy":"not_reported","failure_policy":"publisher treatment","source_ids":["S53","S49"],"fetched_at":"2026-08-31","coverage_state":"reported","notes":"Rank 7."},
  {"result_id":"R14","benchmark_id":"gpqa_diamond","benchmark_version":"198-question Diamond split","task_split":"diamond","model_id":"claude-fable-5","exact_model_version":"not_reported","provider":"Anthropic","evaluator":"Vals AI, read via BenchLM mirror","run_date":"not_reported","published_date":"2026-08-26","as_of":"2026-08-26","prompt":"zero- and few-shot chain-of-thought methods described by evaluator","system_prompt_status":"not_reported","few_shot_configuration":"published headline treatment","reasoning_effort":"not_reported","temperature":"not_reported","max_tokens":"not_reported","tools":"none reported","agent_scaffold":"not_applicable","harness_container":"not_reported","score":93.18,"unit":"percent correct","numerator":"not_reported","denominator":198,"confidence_interval":"not_reported","cost":"not_reported","latency":"not_reported","refusal_policy":"refusal-triggered fallbacks counted as successes","retry_policy":"not_reported","failure_policy":"publisher treatment","source_ids":["S53","S49"],"fetched_at":"2026-08-31","coverage_state":"reported","notes":"Rank 8 under the published treatment. See the paired refusals-as-failures record for the same run."},
  {"result_id":"R15","benchmark_id":"gpqa_diamond","benchmark_version":"198-question Diamond split","task_split":"diamond","model_id":"kimi-k3","exact_model_version":"not_reported","provider":"Moonshot AI","evaluator":"Vals AI, read via BenchLM mirror","run_date":"not_reported","published_date":"2026-08-26","as_of":"2026-08-26","prompt":"zero- and few-shot chain-of-thought methods described by evaluator","system_prompt_status":"not_reported","few_shot_configuration":"published headline treatment","reasoning_effort":"not_reported","temperature":"not_reported","max_tokens":"not_reported","tools":"none reported","agent_scaffold":"not_applicable","harness_container":"not_reported","score":92.93,"unit":"percent correct","numerator":"not_reported","denominator":198,"confidence_interval":"not_reported","cost":"not_reported","latency":"not_reported","refusal_policy":"not_reported","retry_policy":"not_reported","failure_policy":"publisher treatment","source_ids":["S53","S49"],"fetched_at":"2026-08-31","coverage_state":"reported","notes":"Rank 10, tied on the mirror with Grok 4.5."},
  {"result_id":"R16","benchmark_id":"gpqa_diamond","benchmark_version":"198-question Diamond split","task_split":"diamond","model_id":"claude-fable-5","exact_model_version":"not_reported","provider":"Anthropic","evaluator":"Vals AI, read via BenchLM mirror","run_date":"not_reported","published_date":"2026-08-26","as_of":"2026-08-26","prompt":"zero- and few-shot chain-of-thought methods described by evaluator","system_prompt_status":"not_reported","few_shot_configuration":"published headline treatment","reasoning_effort":"not_reported","temperature":"not_reported","max_tokens":"not_reported","tools":"none reported","agent_scaffold":"not_applicable","harness_container":"not_reported","score":55.56,"unit":"percent correct","numerator":"not_reported","denominator":198,"confidence_interval":"not_reported","cost":"not_reported","latency":"not_reported","refusal_policy":"refusals counted as failures; 41.92% refusal rate","retry_policy":"not_reported","failure_policy":"corrected treatment presented by Vals","source_ids":["S49","S13"],"fetched_at":"2026-08-31","coverage_state":"calculated_from_reported_values","notes":"Vals' corrected treatment of the same run, not an independent MTS recalculation. Restated on the 30 August 2026 board. Under this accounting the model moves from fourth of these five to last by a 37-point margin."},
  {"result_id":"R17","benchmark_id":"livecodebench","benchmark_version":"v6","task_split":"overall","model_id":"claude-fable-5","exact_model_version":"not_reported","provider":"Anthropic","evaluator":"Vals AI, read via BenchLM mirror","run_date":"not_reported","published_date":"2026-08-19","as_of":"2026-08-19","prompt":"not_reported","system_prompt_status":"not_reported","few_shot_configuration":"not_reported","reasoning_effort":"not_reported","temperature":"not_reported","max_tokens":"not_reported","tools":"none reported","agent_scaffold":"not_applicable","harness_container":"hidden functional tests","score":89.78,"unit":"percent correct","numerator":"not_reported","denominator":"over 1,000","confidence_interval":"not_reported","cost":"not_reported","latency":"not_reported","refusal_policy":"not_reported","retry_policy":"not_reported","failure_policy":"must pass hidden tests","source_ids":["S54","S50"],"fetched_at":"2026-08-31","coverage_state":"reported","notes":"Mirror leader, consistent with R06."},
  {"result_id":"R18","benchmark_id":"livecodebench","benchmark_version":"v6","task_split":"overall","model_id":"claude-opus-5","exact_model_version":"not_reported","provider":"Anthropic","evaluator":"Vals AI, read via BenchLM mirror","run_date":"not_reported","published_date":"2026-08-19","as_of":"2026-08-19","prompt":"not_reported","system_prompt_status":"not_reported","few_shot_configuration":"not_reported","reasoning_effort":"not_reported","temperature":"not_reported","max_tokens":"not_reported","tools":"none reported","agent_scaffold":"not_applicable","harness_container":"hidden functional tests","score":89.03,"unit":"percent correct","numerator":"not_reported","denominator":"over 1,000","confidence_interval":"not_reported","cost":"not_reported","latency":"not_reported","refusal_policy":"not_reported","retry_policy":"not_reported","failure_policy":"must pass hidden tests","source_ids":["S54","S50"],"fetched_at":"2026-08-31","coverage_state":"reported","notes":"Rank 2, 0.75 points behind."},
  {"result_id":"R19","benchmark_id":"livecodebench","benchmark_version":"v6","task_split":"overall","model_id":"gemini-3.1-pro-preview-02-26","exact_model_version":"not_reported","provider":"Google","evaluator":"Vals AI, read via BenchLM mirror","run_date":"not_reported","published_date":"2026-08-19","as_of":"2026-08-19","prompt":"not_reported","system_prompt_status":"not_reported","few_shot_configuration":"not_reported","reasoning_effort":"not_reported","temperature":"not_reported","max_tokens":"not_reported","tools":"none reported","agent_scaffold":"not_applicable","harness_container":"hidden functional tests","score":88.48,"unit":"percent correct","numerator":"not_reported","denominator":"over 1,000","confidence_interval":"not_reported","cost":"not_reported","latency":"not_reported","refusal_policy":"not_reported","retry_policy":"not_reported","failure_policy":"must pass hidden tests","source_ids":["S54","S50"],"fetched_at":"2026-08-31","coverage_state":"reported","notes":"Rank 4."},
  {"result_id":"R20","benchmark_id":"livecodebench","benchmark_version":"v6","task_split":"overall","model_id":"kimi-k3","exact_model_version":"not_reported","provider":"Moonshot AI","evaluator":"Vals AI, read via BenchLM mirror","run_date":"not_reported","published_date":"2026-08-19","as_of":"2026-08-19","prompt":"not_reported","system_prompt_status":"not_reported","few_shot_configuration":"not_reported","reasoning_effort":"not_reported","temperature":"not_reported","max_tokens":"not_reported","tools":"none reported","agent_scaffold":"not_applicable","harness_container":"hidden functional tests","score":87.19,"unit":"percent correct","numerator":"not_reported","denominator":"over 1,000","confidence_interval":"not_reported","cost":"not_reported","latency":"not_reported","refusal_policy":"not_reported","retry_policy":"not_reported","failure_policy":"must pass hidden tests","source_ids":["S54","S50"],"fetched_at":"2026-08-31","coverage_state":"reported","notes":"Rank 16."},
  {"result_id":"R21","benchmark_id":"livecodebench","benchmark_version":"v6","task_split":"overall","model_id":"gpt-5.6-sol","exact_model_version":"not_reported","provider":"OpenAI","evaluator":"Vals AI, read via BenchLM mirror","run_date":"not_reported","published_date":"2026-08-19","as_of":"2026-08-19","prompt":"not_reported","system_prompt_status":"not_reported","few_shot_configuration":"not_reported","reasoning_effort":"not_reported","temperature":"not_reported","max_tokens":"not_reported","tools":"none reported","agent_scaffold":"not_applicable","harness_container":"hidden functional tests","score":82.6,"unit":"percent correct","numerator":"not_reported","denominator":"over 1,000","confidence_interval":"not_reported","cost":"not_reported","latency":"not_reported","refusal_policy":"not_reported","retry_policy":"not_reported","failure_policy":"must pass hidden tests","source_ids":["S54","S50"],"fetched_at":"2026-08-31","coverage_state":"reported","notes":"Rank 50 of 140 mirrored models. The same model ranks second on GPQA Diamond in this set, the largest rank swing among the five."},
  {"result_id":"R22","benchmark_id":"chatbot_arena","benchmark_version":"Text overall snapshot 2026-08-27","task_split":"overall, no adjustment","model_id":"claude-fable-5","exact_model_version":"claude-fable-5","provider":"Anthropic","evaluator":"Arena","run_date":"rolling through 2026-08-27","published_date":"2026-08-27","as_of":"2026-08-27","prompt":"live user prompts","system_prompt_status":"varies by served model configuration","few_shot_configuration":"not_applicable","reasoning_effort":"model configuration","temperature":"not_reported","max_tokens":"not_reported","tools":"text chat","agent_scaffold":"Arena battle mode","harness_container":"Arena serving stack","score":1507,"unit":"Arena score","numerator":"25,824 model votes","denominator":"7,922,078 total leaderboard votes","confidence_interval":"±5","cost":"not_reported","latency":"not_reported","refusal_policy":"users may select both bad","retry_policy":"continuous battles","failure_policy":"pairwise preference outcome","source_ids":["S51"],"fetched_at":"2026-08-31","coverage_state":"reported","notes":"Raw rank 1. Score moved from 1508 on 21 August 2026 (R07) to 1507 here on a rolling board."},
  {"result_id":"R23","benchmark_id":"chatbot_arena","benchmark_version":"Text overall snapshot 2026-08-27","task_split":"overall, no adjustment","model_id":"claude-opus-5","exact_model_version":"claude-opus-5-high","provider":"Anthropic","evaluator":"Arena","run_date":"rolling through 2026-08-27","published_date":"2026-08-27","as_of":"2026-08-27","prompt":"live user prompts","system_prompt_status":"varies by served model configuration","few_shot_configuration":"not_applicable","reasoning_effort":"high","temperature":"not_reported","max_tokens":"not_reported","tools":"text chat","agent_scaffold":"Arena battle mode","harness_container":"Arena serving stack","score":1492,"unit":"Arena score","numerator":"31,570 model votes","denominator":"7,922,078 total leaderboard votes","confidence_interval":"±5","cost":"not_reported","latency":"not_reported","refusal_policy":"users may select both bad","retry_policy":"continuous battles","failure_policy":"pairwise preference outcome","source_ids":["S51"],"fetched_at":"2026-08-31","coverage_state":"reported","notes":"Raw rank 7. A separate claude-opus-5-max row scores 1488 at rank 12, so this model family occupies two different ranks at once."},
  {"result_id":"R24","benchmark_id":"chatbot_arena","benchmark_version":"Text overall snapshot 2026-08-27","task_split":"overall, no adjustment","model_id":"kimi-k3","exact_model_version":"kimi-k3-max","provider":"Moonshot AI","evaluator":"Arena","run_date":"rolling through 2026-08-27","published_date":"2026-08-27","as_of":"2026-08-27","prompt":"live user prompts","system_prompt_status":"varies by served model configuration","few_shot_configuration":"not_applicable","reasoning_effort":"max","temperature":"not_reported","max_tokens":"not_reported","tools":"text chat","agent_scaffold":"Arena battle mode","harness_container":"Arena serving stack","score":1489,"unit":"Arena score","numerator":"16,586 model votes","denominator":"7,922,078 total leaderboard votes","confidence_interval":"±6","cost":"not_reported","latency":"not_reported","refusal_policy":"users may select both bad","retry_policy":"continuous battles","failure_policy":"pairwise preference outcome","source_ids":["S51"],"fetched_at":"2026-08-31","coverage_state":"reported","notes":"Raw rank 10."},
  {"result_id":"R25","benchmark_id":"chatbot_arena","benchmark_version":"Text overall snapshot 2026-08-27","task_split":"overall, no adjustment","model_id":"gemini-3.1-pro-preview-02-26","exact_model_version":"gemini-3.1-pro-preview","provider":"Google","evaluator":"Arena","run_date":"rolling through 2026-08-27","published_date":"2026-08-27","as_of":"2026-08-27","prompt":"live user prompts","system_prompt_status":"varies by served model configuration","few_shot_configuration":"not_applicable","reasoning_effort":"model configuration","temperature":"not_reported","max_tokens":"not_reported","tools":"text chat","agent_scaffold":"Arena battle mode","harness_container":"Arena serving stack","score":1487,"unit":"Arena score","numerator":"101,163 model votes","denominator":"7,922,078 total leaderboard votes","confidence_interval":"±3","cost":"not_reported","latency":"not_reported","refusal_policy":"users may select both bad","retry_policy":"continuous battles","failure_policy":"pairwise preference outcome","source_ids":["S51"],"fetched_at":"2026-08-31","coverage_state":"reported","notes":"Raw rank 13, with the largest vote count of these five."},
  {"result_id":"R26","benchmark_id":"chatbot_arena","benchmark_version":"Text overall snapshot 2026-08-27","task_split":"overall, no adjustment","model_id":"gpt-5.6-sol","exact_model_version":"gpt-5.6-sol-xhigh","provider":"OpenAI","evaluator":"Arena","run_date":"rolling through 2026-08-27","published_date":"2026-08-27","as_of":"2026-08-27","prompt":"live user prompts","system_prompt_status":"varies by served model configuration","few_shot_configuration":"not_applicable","reasoning_effort":"xhigh","temperature":"not_reported","max_tokens":"not_reported","tools":"text chat","agent_scaffold":"Arena battle mode","harness_container":"Arena serving stack","score":1482,"unit":"Arena score","numerator":"21,537 model votes","denominator":"7,922,078 total leaderboard votes","confidence_interval":"±5","cost":"not_reported","latency":"not_reported","refusal_policy":"users may select both bad","retry_policy":"continuous battles","failure_policy":"pairwise preference outcome","source_ids":["S51"],"fetched_at":"2026-08-31","coverage_state":"reported","notes":"Raw rank 17. Confidence intervals in this band overlap several adjacent ranks."},
  {"result_id":"R27","benchmark_id":"aa_intelligence_index","benchmark_version":"v4.1.1","task_split":"nine-evaluation weighted composite","model_id":"claude-opus-5","exact_model_version":"Claude Opus 5 (Max Effort)","provider":"Anthropic","evaluator":"Artificial Analysis","run_date":"not_reported","published_date":"not_reported","as_of":"2026-08-31","prompt":"component prompts described by evaluator; exact suite prompt set partially reported","system_prompt_status":"component-specific; not fully reported","few_shot_configuration":"varies by component evaluation","reasoning_effort":"Adaptive Reasoning, Max Effort","temperature":"not_reported","max_tokens":"not_reported","tools":"varies by component; see evaluator methodology","agent_scaffold":"Artificial Analysis evaluation harness","harness_container":"not_reported","score":63,"unit":"0-100 weighted Intelligence Index","numerator":"not_applicable","denominator":"nine evaluations","confidence_interval":"index-level 95% confidence interval estimated below ±1 point","cost":"not_reported","latency":"not_reported","refusal_policy":"component-specific; not_reported in composite record","retry_policy":"component-specific repeats published in methodology","failure_policy":"component-specific scoring aggregated by published category weights","source_ids":["S52","S38","S39"],"fetched_at":"2026-08-31","coverage_state":"reported","notes":"Rank 1 on the accessed leaderboard, consistent with R08. A Claude Opus 5 (xhigh) row also scores 63; a (high) row scores 61 and a (medium) row 59, so one model family spans four index values."},
  {"result_id":"R28","benchmark_id":"aa_intelligence_index","benchmark_version":"v4.1.1","task_split":"nine-evaluation weighted composite","model_id":"claude-fable-5","exact_model_version":"Claude Fable 5 (with fallback)","provider":"Anthropic","evaluator":"Artificial Analysis","run_date":"not_reported","published_date":"not_reported","as_of":"2026-08-31","prompt":"component prompts described by evaluator; exact suite prompt set partially reported","system_prompt_status":"component-specific; not fully reported","few_shot_configuration":"varies by component evaluation","reasoning_effort":"not_reported","temperature":"not_reported","max_tokens":"not_reported","tools":"varies by component; see evaluator methodology","agent_scaffold":"Artificial Analysis evaluation harness","harness_container":"not_reported","score":62,"unit":"0-100 weighted Intelligence Index","numerator":"not_applicable","denominator":"nine evaluations","confidence_interval":"index-level 95% confidence interval estimated below ±1 point","cost":"not_reported","latency":"not_reported","refusal_policy":"component-specific; not_reported in composite record","retry_policy":"component-specific repeats published in methodology","failure_policy":"component-specific scoring aggregated by published category weights","source_ids":["S52","S38","S39"],"fetched_at":"2026-08-31","coverage_state":"reported","notes":"Rank 3. The evaluator labels this configuration as using a fallback."},
  {"result_id":"R29","benchmark_id":"aa_intelligence_index","benchmark_version":"v4.1.1","task_split":"nine-evaluation weighted composite","model_id":"gpt-5.6-sol","exact_model_version":"GPT-5.6 Sol (Max Effort)","provider":"OpenAI","evaluator":"Artificial Analysis","run_date":"not_reported","published_date":"not_reported","as_of":"2026-08-31","prompt":"component prompts described by evaluator; exact suite prompt set partially reported","system_prompt_status":"component-specific; not fully reported","few_shot_configuration":"varies by component evaluation","reasoning_effort":"max","temperature":"not_reported","max_tokens":"not_reported","tools":"varies by component; see evaluator methodology","agent_scaffold":"Artificial Analysis evaluation harness","harness_container":"not_reported","score":61,"unit":"0-100 weighted Intelligence Index","numerator":"not_applicable","denominator":"nine evaluations","confidence_interval":"index-level 95% confidence interval estimated below ±1 point","cost":"not_reported","latency":"not_reported","refusal_policy":"component-specific; not_reported in composite record","retry_policy":"component-specific repeats published in methodology","failure_policy":"component-specific scoring aggregated by published category weights","source_ids":["S52","S38","S39"],"fetched_at":"2026-08-31","coverage_state":"reported","notes":"Rank 5. Separate xhigh and high rows score 59 and 57."},
  {"result_id":"R30","benchmark_id":"aa_intelligence_index","benchmark_version":"v4.1.1","task_split":"nine-evaluation weighted composite","model_id":"kimi-k3","exact_model_version":"Kimi K3 (Max Effort)","provider":"Moonshot AI","evaluator":"Artificial Analysis","run_date":"not_reported","published_date":"not_reported","as_of":"2026-08-31","prompt":"component prompts described by evaluator; exact suite prompt set partially reported","system_prompt_status":"component-specific; not fully reported","few_shot_configuration":"varies by component evaluation","reasoning_effort":"max","temperature":"not_reported","max_tokens":"not_reported","tools":"varies by component; see evaluator methodology","agent_scaffold":"Artificial Analysis evaluation harness","harness_container":"not_reported","score":60,"unit":"0-100 weighted Intelligence Index","numerator":"not_applicable","denominator":"nine evaluations","confidence_interval":"index-level 95% confidence interval estimated below ±1 point","cost":"not_reported","latency":"not_reported","refusal_policy":"component-specific; not_reported in composite record","retry_policy":"component-specific repeats published in methodology","failure_policy":"component-specific scoring aggregated by published category weights","source_ids":["S52","S38","S39"],"fetched_at":"2026-08-31","coverage_state":"reported","notes":"Rank 8."},
  {"result_id":"R31","benchmark_id":"swebench_verified","benchmark_version":"500-task verified split","task_split":"verified","model_id":"gpt-5.6-sol","exact_model_version":"not_reported","provider":"OpenAI","evaluator":"Vals AI","run_date":"not_reported","published_date":"2026-08-30","as_of":"2026-08-30","prompt":"system prompt described by evaluator; exact text not_reported","system_prompt_status":"described, exact text not_reported","few_shot_configuration":"not_applicable","reasoning_effort":"not_reported","temperature":"not_reported","max_tokens":"not_reported","tools":"bash only","agent_scaffold":"minimal bash-tool-only agent harness","harness_container":"isolated SWE-bench Docker task containers","score":96.2,"unit":"percent resolved","numerator":"not_reported","denominator":500,"confidence_interval":"not_reported","cost":"not_reported","latency":"not_reported","refusal_policy":"not_reported","retry_policy":"not_reported","failure_policy":"all required tests must pass","source_ids":["S48"],"fetched_at":"2026-09-01","coverage_state":"reported","notes":"Sourced Sol result used in the opening. Exact model checkpoint, run date, and reasoning effort were not reported."}
]
