{"schema":"postcutoff/itemlist@1","as_of":"2026-10-10T14:45:00+02:00","url":"https://postcutoff.com/news/benchmark/","md":"https://postcutoff.com/news/benchmark/index.md","disclosure":{"written_by":"AI agents (Claude Opus 5.5 in Claude Code)","editor":"Adam Bicz","policy":"https://postcutoff.com/about/"},"license":null,"item_type":"Event","scope":"benchmark","page":1,"pages":1,"total":24,"per_page":50,"feed":"https://postcutoff.com/feeds/benchmark.xml","items":[{"id":"2026-10-08-arena-200m-3-1b-alignment-index","url":"https://postcutoff.com/e/2026-10-08-arena-200m-3-1b-alignment-index/","date":"2026-10-08","date_precision":"day","short_title":"Arena raises $200M at $3.1B and launches an AI Alignment Index for agents","deck":"OpenAI models lead, Claude Opus 5.5 is 6th","takeaway":"On Oct 8, 2026 Arena, the crowdsourced model-leaderboard company that began as UC Berkeley's Chatbot Arena, raised a $200M Series B at a $3.1B valuation (up from $1.7B in January), led by Lightspeed and Khosla.","category":"benchmark","category_label":"Benchmarks","importance":3,"confidence":"high","status":{"key":"confirmed","labels":["Confirmed"]},"sources":4,"official":2,"filed":"2026-10-08","updated":"2026-10-08","orgs":["Arena"]},{"id":"2026-10-08-openproblembench-82-open-problems","url":"https://postcutoff.com/e/2026-10-08-openproblembench-82-open-problems/","date":"2026-10-08","date_precision":"day","short_title":"OpenProblemBench: 82 unresolved math and theoretical-physics problems","deck":"GPT-6 Astra has the highest model-judged solve rate (14%)","takeaway":"It adds a benchmark of open research problems, after FrontierMath-style tests with known answers.","category":"benchmark","category_label":"Benchmarks","importance":2,"confidence":"high","status":{"key":"confirmed","labels":["Confirmed"]},"sources":1,"official":1,"filed":"2026-10-09","updated":"2026-10-09","orgs":["Institute of Theoretical Physics CAS","University of Science and Technology of China"]},{"id":"2026-10-01-arc-prize-2026-milestone-2-winners","url":"https://postcutoff.com/e/2026-10-01-arc-prize-2026-milestone-2-winners/","date":"2026-10-01","date_precision":"day","short_title":"Open-source ARC-AGI-3 agents top out at 27.9% at ARC Prize 2026 Milestone #2","deck":null,"takeaway":"The Kaggle track runs open models under Kaggle's compute limits, so it measures how far open methods get without frontier APIs.","category":"benchmark","category_label":"Benchmarks","importance":2,"confidence":"high","status":{"key":"confirmed","labels":["Confirmed"]},"sources":8,"official":8,"filed":"2026-10-04","updated":"2026-10-04","orgs":["ARC Prize Foundation","Tufa Labs"]},{"id":"2026-09-28-artificial-analysis-cyber-index","url":"https://postcutoff.com/e/2026-09-28-artificial-analysis-cyber-index/","date":"2026-09-28","date_precision":"day","short_title":"Artificial Analysis launches the Cyber Index and an industry alliance (IBM, NVIDIA, Vercel, Collinear) for AI vulnerability-fixing evals","deck":null,"takeaway":"Cyber capability is now the main axis of frontier-model risk debates.","category":"benchmark","category_label":"Benchmarks","importance":2,"confidence":"high","status":{"key":"confirmed","labels":["Confirmed"]},"sources":1,"official":1,"filed":"2026-09-30","updated":"2026-09-30","orgs":["Artificial Analysis","IBM","NVIDIA","Vercel","Collinear AI"]},{"id":"2026-09-24-c5r-facility-0-sciuniverse-benchmark","url":"https://postcutoff.com/e/2026-09-24-c5r-facility-0-sciuniverse-benchmark/","date":"2026-09-24","date_precision":"day","short_title":"C5R opens Facility-0, an AI-run wet lab, and the SciUniverse benchmark","deck":"Claude Fable 5.1 leads with 45.3%","takeaway":"Most AI-for-science benchmarks test reasoning on text or data.","category":"benchmark","category_label":"Benchmarks","importance":3,"confidence":"high","status":{"key":"confirmed","labels":["Confirmed"]},"sources":6,"official":2,"filed":"2026-10-05","updated":"2026-10-05","orgs":["C5R"]},{"id":"2026-09-23-openai-mentalhealthbench","url":"https://postcutoff.com/e/2026-09-23-openai-mentalhealthbench/","date":"2026-09-23","date_precision":"day","short_title":"OpenAI releases MentalHealthBench, an open benchmark for AI in mental-health conversations","deck":null,"takeaway":"Mental-health use of chatbots was a major 2025–2026 safety and litigation topic, and this gives labs and regulators a shared measure.","category":"benchmark","category_label":"Benchmarks","importance":3,"confidence":"medium","status":{"key":"partly","labels":["Partly confirmed"]},"sources":4,"official":1,"filed":"2026-09-29","updated":"2026-09-29","orgs":["OpenAI"]},{"id":"2026-09-23-drivingbench-llms-drive-real-car","url":"https://postcutoff.com/e/2026-09-23-drivingbench-llms-drive-real-car/","date":"2026-09-23","date_precision":"day","short_title":"DrivingBench: GPT-6 Astra is the first frontier LLM to complete a cone course driving a real Toyota Corolla","deck":null,"takeaway":"A small but vivid probe of general models' embodied, closed-loop control.","category":"benchmark","category_label":"Benchmarks","importance":2,"confidence":"medium","status":{"key":"partly","labels":["Partly confirmed"]},"sources":2,"official":1,"filed":"2026-09-30","updated":"2026-09-30","orgs":["OpenAI"]},{"id":"2026-09-22-hle-diamond","url":"https://postcutoff.com/e/2026-09-22-hle-diamond/","date":"2026-09-22","date_precision":"day","short_title":"CAIS releases HLE-Diamond, a refined 1,000-question Humanity's Last Exam","deck":"GPT-6 Astra scores 82.9% with tools","takeaway":"If confirmed, the benchmark once billed as 'the last exam' is close to saturation within about 20 months of release.","category":"benchmark","category_label":"Benchmarks","importance":3,"confidence":"high","status":{"key":"confirmed","labels":["Confirmed"]},"sources":3,"official":2,"filed":"2026-09-29","updated":"2026-09-30","orgs":["Center for AI Safety","Scale AI"]},{"id":"2026-09-16-mlperf-inference-v6-1","url":"https://postcutoff.com/e/2026-09-16-mlperf-inference-v6-1/","date":"2026-09-16","date_precision":"day","short_title":"MLPerf Inference v6.1 draws a record 30 submitters and the first Vera Rubin NVL72 results","deck":null,"takeaway":"MLPerf is the main audited cross-vendor inference benchmark.","category":"benchmark","category_label":"Benchmarks","importance":3,"confidence":"high","status":{"key":"confirmed","labels":["Confirmed"]},"sources":7,"official":5,"filed":"2026-10-02","updated":"2026-10-02","orgs":["MLCommons","NVIDIA"]},{"id":"2026-09-03-arc-agi-3-gpt-6-astra","url":"https://postcutoff.com/e/2026-09-03-arc-agi-3-gpt-6-astra/","date":"2026-09-03","date_precision":"day","short_title":"GPT-6 Astra scores 62.7% on ARC-AGI-3, outacting humans on 96% of levels","deck":null,"takeaway":"ARC-AGI-3 was meant to measure human-like skill acquisition; its near-saturation (and the harness gap) shows both how fast agentic reasoning improved in 2026 and how much scaffolding now drives scores.","category":"benchmark","category_label":"Benchmarks","importance":5,"confidence":"high","status":{"key":"confirmed","labels":["Confirmed"]},"sources":7,"official":4,"filed":"2026-09-29","updated":"2026-10-04","orgs":["ARC Prize Foundation","OpenAI"]},{"id":"2026-09-03-qwen-e-commerce-bench","url":"https://postcutoff.com/e/2026-09-03-qwen-e-commerce-bench/","date":"2026-09-03","date_precision":"day","short_title":"Qwen and Taobao release E-Commerce Bench","deck":"18 models run online stores for a simulated year","takeaway":"Results differed by orders of magnitude, and the top earner was not the most careful operator.","category":"benchmark","category_label":"Benchmarks","importance":2,"confidence":"high","status":{"key":"confirmed","labels":["Confirmed"]},"sources":4,"official":4,"filed":"2026-09-30","updated":"2026-09-30","orgs":["Alibaba","Qwen","Taobao & Tmall Group"]},{"id":"2026-08-26-bixbench3-computational-biology-agents","url":"https://postcutoff.com/e/2026-08-26-bixbench3-computational-biology-agents/","date":"2026-08-26","date_precision":"day","short_title":"BixBench3 tests AI agents on whole computational-biology studies from raw data","deck":"Best score 0.48 (GPT-5.6 Sol)","takeaway":"It is an end-to-end test for \"AI scientist\" claims in biology.","category":"benchmark","category_label":"Benchmarks","importance":2,"confidence":"high","status":{"key":"confirmed","labels":["Confirmed"]},"sources":5,"official":4,"filed":"2026-10-04","updated":"2026-10-04","orgs":["Edison Scientific"]},{"id":"2026-08-24-artificial-analysis-speech-agent-arena","url":"https://postcutoff.com/e/2026-08-24-artificial-analysis-speech-agent-arena/","date":"2026-08-24","date_precision":"day","short_title":"Artificial Analysis launches the Speech Agent Arena for speech-to-speech voice agents","deck":null,"takeaway":"Voice agents are being sold into customer service, where finishing the task matters more than sounding natural.","category":"benchmark","category_label":"Benchmarks","importance":2,"confidence":"high","status":{"key":"confirmed","labels":["Confirmed"]},"sources":4,"official":4,"filed":"2026-09-29","updated":"2026-09-29","orgs":["Artificial Analysis"]},{"id":"2026-07-24-anthropic-andon-project-pilot-drone-bench","url":"https://postcutoff.com/e/2026-07-24-anthropic-andon-project-pilot-drone-bench/","date":"2026-07-24","date_precision":"day","short_title":"Project Pilot: Anthropic and Andon Labs test whether AI models can fly a surveillance drone (Drone-Bench)","deck":null,"takeaway":"Autonomous aerial surveillance is a clearly dual-use capability, and the study suggests frontier models were close to it by mid-2026, except for 3D mapping.","category":"benchmark","category_label":"Benchmarks","importance":3,"confidence":"high","status":{"key":"confirmed","labels":["Confirmed"]},"sources":5,"official":3,"filed":"2026-10-01","updated":"2026-10-01","orgs":["Anthropic","Andon Labs"]},{"id":"2026-06-17-openai-lifescibench","url":"https://postcutoff.com/e/2026-06-17-openai-lifescibench/","date":"2026-06-17","date_precision":"day","short_title":"OpenAI releases LifeSciBench, 750 expert-written life-science research tasks","deck":"GPT-Rosalind leads with a 36% pass rate","takeaway":"It is a large, expert-graded measure of AI for biology research.","category":"benchmark","category_label":"Benchmarks","importance":2,"confidence":"medium","status":{"key":"partly","labels":["Partly confirmed"]},"sources":4,"official":1,"filed":"2026-10-04","updated":"2026-10-04","orgs":["OpenAI"]},{"id":"2026-06-10-first-proof-second-batch","url":"https://postcutoff.com/e/2026-06-10-first-proof-second-batch/","date":"2026-06-10","date_precision":"day","short_title":"AI systems pass 7 of 10 unpublished research problems in First Proof's second batch","deck":null,"takeaway":"It is the most carefully refereed measurement so far of AI on genuine research problems.","category":"benchmark","category_label":"Benchmarks","importance":3,"confidence":"high","status":{"key":"confirmed","labels":["Confirmed"]},"sources":6,"official":4,"filed":"2026-10-05","updated":"2026-10-05","orgs":["First Proof Foundation","ETH Zurich","OpenAI","UCLA","Princeton University"]},{"id":"2026-03-25-arc-agi-3-launch","url":"https://postcutoff.com/e/2026-03-25-arc-agi-3-launch/","date":"2026-03-25","date_precision":"day","short_title":"ARC Prize launches ARC-AGI-3, an interactive game benchmark where frontier AI scored under 1%","deck":null,"takeaway":"It was designed as the hardest-to-game AGI benchmark of 2026; within six months it was largely cracked (see GPT-6 Astra entry), illustrating the pace of agentic progress.","category":"benchmark","category_label":"Benchmarks","importance":4,"confidence":"medium","status":{"key":"partly","labels":["Partly confirmed"]},"sources":4,"official":3,"filed":"2026-09-29","updated":"2026-09-29","orgs":["ARC Prize Foundation"]},{"id":"2026-02-11-ai2-molmospaces","url":"https://postcutoff.com/e/2026-02-11-ai2-molmospaces/","date":"2026-02-11","date_precision":"day","short_title":"Ai2 launches MolmoSpaces, an open simulation ecosystem and leaderboard for generalist robot policies","deck":null,"takeaway":"Robot learning lacked shared benchmarks like those language models have.","category":"benchmark","category_label":"Benchmarks","importance":2,"confidence":"high","status":{"key":"confirmed","labels":["Confirmed"]},"sources":3,"official":3,"filed":"2026-09-29","updated":"2026-09-29","orgs":["Ai2"]},{"id":"2026-01-29-metr-time-horizon-1-1","url":"https://postcutoff.com/e/2026-01-29-metr-time-horizon-1-1/","date":"2026-01-29","date_precision":"day","short_title":"METR releases Time Horizon 1.1 with expanded long-task suite","deck":null,"takeaway":"As frontier horizons approach the top of the suite, METR's own caveat (unreliable >16h) signals the benchmark itself is near saturation.","category":"benchmark","category_label":"Benchmarks","importance":3,"confidence":"high","status":{"key":"confirmed","labels":["Confirmed"]},"sources":3,"official":3,"filed":"2026-09-29","updated":"2026-09-29","orgs":["METR"]},{"id":"2025-12-06-virtual-cell-challenge-2025-winners","url":"https://postcutoff.com/e/2025-12-06-virtual-cell-challenge-2025-winners/","date":"2025-12-06","date_precision":"day","short_title":"Arc Institute announces first Virtual Cell Challenge winners","deck":"A 2026 zero-shot round follows","takeaway":"Virtual cells are a major goal of AI biology, and this challenge is becoming the field's shared benchmark.","category":"benchmark","category_label":"Benchmarks","importance":2,"confidence":"high","status":{"key":"confirmed","labels":["Confirmed"]},"sources":4,"official":4,"filed":"2026-09-29","updated":"2026-09-29","orgs":["Arc Institute","BioMap","Altos Labs","NVIDIA"]},{"id":"2025-09-17-icpc-gold-ai","url":"https://postcutoff.com/e/2025-09-17-icpc-gold-ai/","date":"2025-09-17","date_precision":"day","short_title":"AI reaches gold-medal level at the ICPC World Finals","deck":null,"takeaway":"Following IMO gold, confirmed elite-human-level algorithmic problem solving by general-purpose reasoning models.","category":"benchmark","category_label":"Benchmarks","importance":4,"confidence":"medium","status":{"key":"confirmed","labels":["Result confirmed"]},"sources":2,"official":1,"filed":"2026-09-29","updated":"2026-09-29","orgs":["OpenAI","Google DeepMind"]},{"id":"2025-06-22-roboarena","url":"https://postcutoff.com/e/2025-06-22-roboarena/","date":"2025-06-22","date_precision":"day","short_title":"RoboArena: crowd-sourced, double-blind real-world evaluation of generalist robot policies","deck":null,"takeaway":"Real-world robot evaluation is expensive and hard to standardize.","category":"benchmark","category_label":"Benchmarks","importance":2,"confidence":"high","status":{"key":"confirmed","labels":["Confirmed"]},"sources":2,"official":2,"filed":"2026-09-29","updated":"2026-09-29","orgs":["RoboArena consortium"]},{"id":"2024-12-20-openai-o3","url":"https://postcutoff.com/e/2024-12-20-openai-o3/","date":"2024-12-20","date_precision":"day","short_title":"OpenAI announces o3, scoring 75.7–87.5% on ARC-AGI","deck":null,"takeaway":"Convinced many observers that reasoning models were on a steep trajectory; ARC Prize called it a genuine step-change.","category":"benchmark","category_label":"Benchmarks","importance":5,"confidence":"high","status":{"key":"confirmed","labels":["Confirmed"]},"sources":2,"official":2,"filed":"2026-09-29","updated":"2026-09-29","orgs":["OpenAI","ARC Prize"]},{"id":"2009-06-20-imagenet","url":"https://postcutoff.com/e/2009-06-20-imagenet/","date":"2009-06-20","date_precision":"month","short_title":"ImageNet dataset presented at CVPR 2009","deck":null,"takeaway":"Showed that data scale was a key ingredient of progress; AlexNet's 2012 ImageNet win kicked off the deep learning era.","category":"benchmark","category_label":"Benchmarks","importance":5,"confidence":"high","status":{"key":"confirmed","labels":["Confirmed"]},"sources":3,"official":2,"filed":"2026-09-29","updated":"2026-09-29","orgs":["Princeton University","Stanford University"]}]}