{"catalog":{"title":"LLM leaderboard: benchmark scores from labs and maintainers","path":"/leaderboard","updatedAt":"2026-10-05T05:11:35.668Z","fields":[{"id":"model","label":"Model","type":"text"},{"id":"lab","label":"Lab","type":"select","options":["OpenAI","Anthropic","Google DeepMind","xAI","Moonshot AI","Meta","DeepSeek","Alibaba Qwen","Mistral AI"]},{"id":"benchmark","label":"Benchmark","type":"select","options":["Artificial Analysis Intelligence Index","ARC-AGI-3","AutomationBench","AutomationBench-AA","BrowseComp","CursorBench 4.0","DeepSWE v1.1","FrontierMath Tier 4","GDPval-AA v2.1","GPQA Diamond","HLE-Verified","Humanity's Last Exam","LVBench","OSWorld 2.0","OSWorld 2.1","SciCode","SWE-bench Pro","SWE-bench Verified","Terminal-Bench 2.1","Terminal-Bench 4.0"]},{"id":"score","label":"Score","type":"number","format":{"maximumFractionDigits":1}},{"id":"unit","label":"Unit","type":"select","options":["%","Elo","Index points"]},{"id":"setting","label":"Setting","type":"text"},{"id":"reportedBy","label":"Reported by","type":"select","options":["Lab (self-reported)","Benchmark maintainer","Independent evaluator"]},{"id":"publisher","label":"Publisher","type":"text"},{"id":"published","label":"Published","type":"date"},{"id":"source","label":"Source","type":"url"},{"id":"checked","label":"Checked","type":"date"}]},"entries":[{"slug":"gpt-6-astra-high-arc-agi-3-provider-adapter-harness","name":"GPT-6 Astra (high) — ARC-AGI-3 (ARC Prize, Provider Adapter harness)","path":null,"category":null,"updatedAt":"2026-10-05T05:10:40.558Z","fields":{"model":"GPT-6 Astra (high)","lab":"OpenAI","benchmark":"ARC-AGI-3","score":99.9,"unit":"%","setting":"High effort; Provider Adapter harness; Semi-Private set","reportedBy":"Benchmark maintainer","publisher":"ARC Prize Foundation","published":"2026-09-03","source":"https://arcprize.org/blog/astra","checked":"2026-10-05"}},{"slug":"gpt-6-astra-max-arc-agi-3-standard-harness","name":"GPT-6 Astra (max) — ARC-AGI-3 (ARC Prize, Standard harness)","path":null,"category":null,"updatedAt":"2026-10-05T05:10:43.776Z","fields":{"model":"GPT-6 Astra (max)","lab":"OpenAI","benchmark":"ARC-AGI-3","score":62.7,"unit":"%","setting":"Max effort; Standard harness; Semi-Private set","reportedBy":"Benchmark maintainer","publisher":"ARC Prize Foundation","published":"2026-09-03","source":"https://arcprize.org/blog/astra","checked":"2026-10-05"}},{"slug":"claude-opus-5-5-max-artificial-analysis-intelligence-index","name":"Claude Opus 5.5 (max) — Artificial Analysis Intelligence Index (Sept 22 article)","path":null,"category":null,"updatedAt":"2026-10-05T05:09:56.129Z","fields":{"model":"Claude Opus 5.5 (max)","lab":"Anthropic","benchmark":"Artificial Analysis Intelligence Index","score":58,"unit":"Index points","setting":"Max effort; index as of AA article dated 2026-09-22","reportedBy":"Independent evaluator","publisher":"Artificial Analysis","published":"2026-09-22","source":"https://artificialanalysis.ai/articles/claude-opus-5-5","checked":"2026-10-05"}},{"slug":"gemini-3-8-flash-high-artificial-analysis-intelligence-index","name":"Gemini 3.8 Flash (high) — Artificial Analysis Intelligence Index (Sept 30 article)","path":null,"category":null,"updatedAt":"2026-10-05T05:10:53.256Z","fields":{"model":"Gemini 3.8 Flash (high)","lab":"Google DeepMind","benchmark":"Artificial Analysis Intelligence Index","score":41,"unit":"Index points","setting":"High reasoning; index as of AA article dated 2026-09-30","reportedBy":"Independent evaluator","publisher":"Artificial Analysis","published":"2026-09-30","source":"https://artificialanalysis.ai/articles/gemini-4-argon-google-top-three-labs","checked":"2026-10-05"}},{"slug":"gemini-4-argon-high-artificial-analysis-intelligence-index","name":"Gemini 4 Argon (high) — Artificial Analysis Intelligence Index (Sept 30 article)","path":null,"category":null,"updatedAt":"2026-10-05T05:11:00.977Z","fields":{"model":"Gemini 4 Argon (high)","lab":"Google DeepMind","benchmark":"Artificial Analysis Intelligence Index","score":53,"unit":"Index points","setting":"High reasoning; index as of AA article dated 2026-09-30","reportedBy":"Independent evaluator","publisher":"Artificial Analysis","published":"2026-09-30","source":"https://artificialanalysis.ai/articles/gemini-4-argon-google-top-three-labs","checked":"2026-10-05"}},{"slug":"gpt-6-astra-max-artificial-analysis-intelligence-index","name":"GPT-6 Astra (max) — Artificial Analysis Intelligence Index (Sept 30 article)","path":null,"category":null,"updatedAt":"2026-10-05T05:10:47.114Z","fields":{"model":"GPT-6 Astra (max)","lab":"OpenAI","benchmark":"Artificial Analysis Intelligence Index","score":53,"unit":"Index points","setting":"Max effort; index as of AA article dated 2026-09-30","reportedBy":"Independent evaluator","publisher":"Artificial Analysis","published":"2026-09-30","source":"https://artificialanalysis.ai/articles/gemini-4-argon-google-top-three-labs","checked":"2026-10-05"}},{"slug":"gpt-6-1-sol-max-artificial-analysis-intelligence-index","name":"GPT-6.1 Sol (max) — Artificial Analysis Intelligence Index (Sept 30 article)","path":null,"category":null,"updatedAt":"2026-10-05T05:10:49.907Z","fields":{"model":"GPT-6.1 Sol (max)","lab":"OpenAI","benchmark":"Artificial Analysis Intelligence Index","score":52,"unit":"Index points","setting":"Max effort; index as of AA article dated 2026-09-30","reportedBy":"Independent evaluator","publisher":"Artificial Analysis","published":"2026-09-30","source":"https://artificialanalysis.ai/articles/gemini-4-argon-google-top-three-labs","checked":"2026-10-05"}},{"slug":"claude-opus-5-5-automationbench-anthropic","name":"Claude Opus 5.5 — AutomationBench (Anthropic, self-reported)","path":null,"category":null,"updatedAt":"2026-10-05T05:10:13.735Z","fields":{"model":"Claude Opus 5.5","lab":"Anthropic","benchmark":"AutomationBench","score":40,"unit":"%","setting":"As reported in Anthropic's table","reportedBy":"Lab (self-reported)","publisher":"Anthropic","published":"2026-09-22","source":"https://www.anthropic.com/claude-opus-5-5","checked":"2026-10-05"}},{"slug":"gemini-4-argon-automationbench-google","name":"Gemini 4 Argon — AutomationBench (Google, self-reported)","path":null,"category":null,"updatedAt":"2026-10-05T05:11:02.761Z","fields":{"model":"Gemini 4 Argon","lab":"Google DeepMind","benchmark":"AutomationBench","score":51.3,"unit":"%","setting":"As reported in Google's announcement","reportedBy":"Lab (self-reported)","publisher":"Google","published":"2026-09-30","source":"https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-4-argon/","checked":"2026-10-05"}},{"slug":"claude-sonnet-5-5-max-automationbench-aa-artificial-analysis","name":"Claude Sonnet 5.5 (max) — AutomationBench-AA (Artificial Analysis)","path":null,"category":null,"updatedAt":"2026-10-05T05:10:25.663Z","fields":{"model":"Claude Sonnet 5.5 (max)","lab":"Anthropic","benchmark":"AutomationBench-AA","score":71,"unit":"%","setting":"Max effort (Artificial Analysis run)","reportedBy":"Independent evaluator","publisher":"Artificial Analysis","published":"2026-09-30","source":"https://artificialanalysis.ai/articles/gemini-4-argon-google-top-three-labs","checked":"2026-10-05"}},{"slug":"gemini-4-argon-automationbench-aa-artificial-analysis","name":"Gemini 4 Argon — AutomationBench-AA (Artificial Analysis)","path":null,"category":null,"updatedAt":"2026-10-05T05:11:05.631Z","fields":{"model":"Gemini 4 Argon","lab":"Google DeepMind","benchmark":"AutomationBench-AA","score":78,"unit":"%","setting":"Artificial Analysis run","reportedBy":"Independent evaluator","publisher":"Artificial Analysis","published":"2026-09-30","source":"https://artificialanalysis.ai/articles/gemini-4-argon-google-top-three-labs","checked":"2026-10-05"}},{"slug":"gpt-5-6-sol-browsecomp-openai","name":"GPT-5.6 Sol — BrowseComp (OpenAI, self-reported)","path":null,"category":null,"updatedAt":"2026-10-02T12:12:24.928Z","fields":{"model":"GPT-5.6 Sol","lab":"OpenAI","benchmark":"BrowseComp","score":90.4,"unit":"%","setting":"As reported in OpenAI's announcement","reportedBy":"Lab (self-reported)","publisher":"OpenAI","published":"2026-07-09","source":"https://openai.com/index/gpt-5-6/","checked":"2026-10-02"}},{"slug":"gpt-6-astra-browsecomp-openai","name":"GPT-6 Astra — BrowseComp (OpenAI, self-reported)","path":null,"category":null,"updatedAt":"2026-10-02T12:13:16.465Z","fields":{"model":"GPT-6 Astra","lab":"OpenAI","benchmark":"BrowseComp","score":91.5,"unit":"%","setting":"Maximum across effort levels","reportedBy":"Lab (self-reported)","publisher":"OpenAI","published":"2026-09-22","source":"https://openai.com/index/gpt-6-astra/","checked":"2026-10-02"}},{"slug":"kimi-k3-max-browsecomp-moonshot","name":"Kimi K3 (max) — BrowseComp (Moonshot AI, 1M-token context)","path":null,"category":null,"updatedAt":"2026-10-05T05:11:26.319Z","fields":{"model":"Kimi K3 (max)","lab":"Moonshot AI","benchmark":"BrowseComp","score":90.4,"unit":"%","setting":"Max reasoning effort; 1M-token context, no context management","reportedBy":"Lab (self-reported)","publisher":"Moonshot AI","published":"2026-07-17","source":"https://www.kimi.ai/blog/kimi-k3","checked":"2026-10-05"}},{"slug":"claude-opus-5-5-cursorbench-4-anthropic","name":"Claude Opus 5.5 — CursorBench 4.0 (Anthropic, self-reported)","path":null,"category":null,"updatedAt":"2026-10-05T05:10:16.482Z","fields":{"model":"Claude Opus 5.5","lab":"Anthropic","benchmark":"CursorBench 4.0","score":57.8,"unit":"%","setting":"As reported in Anthropic's table","reportedBy":"Lab (self-reported)","publisher":"Anthropic","published":"2026-09-22","source":"https://www.anthropic.com/claude-opus-5-5","checked":"2026-10-05"}},{"slug":"claude-sonnet-5-5-cursorbench-4-anthropic","name":"Claude Sonnet 5.5 — CursorBench 4.0 (Anthropic, self-reported)","path":null,"category":null,"updatedAt":"2026-10-05T05:10:31.951Z","fields":{"model":"Claude Sonnet 5.5","lab":"Anthropic","benchmark":"CursorBench 4.0","score":55.5,"unit":"%","setting":"As reported in Anthropic's table","reportedBy":"Lab (self-reported)","publisher":"Anthropic","published":"2026-09-28","source":"https://www.anthropic.com/claude-sonnet-5-5","checked":"2026-10-05"}},{"slug":"grok-4-7-cursorbench-4-xai","name":"Grok 4.7 — CursorBench 4.0 (xAI, self-reported)","path":null,"category":null,"updatedAt":"2026-10-05T05:11:20.174Z","fields":{"model":"Grok 4.7","lab":"xAI","benchmark":"CursorBench 4.0","score":46.3,"unit":"%","setting":"As reported in xAI's announcement","reportedBy":"Lab (self-reported)","publisher":"xAI","published":"2026-09-21","source":"https://x.ai/news/grok-4-7","checked":"2026-10-05"}},{"slug":"gemini-4-argon-deepswe-v1-1-google","name":"Gemini 4 Argon — DeepSWE v1.1 (Google, self-reported)","path":null,"category":null,"updatedAt":"2026-10-05T05:11:08.515Z","fields":{"model":"Gemini 4 Argon","lab":"Google DeepMind","benchmark":"DeepSWE v1.1","score":77.9,"unit":"%","setting":"As reported in Google's announcement","reportedBy":"Lab (self-reported)","publisher":"Google","published":"2026-09-30","source":"https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-4-argon/","checked":"2026-10-05"}},{"slug":"gpt-6-luna-max-deepswe-v1-1-openai","name":"GPT-6 Luna (max) — DeepSWE v1.1 (OpenAI, self-reported)","path":null,"category":null,"updatedAt":"2026-10-02T12:09:05.401Z","fields":{"model":"GPT-6 Luna (max)","lab":"OpenAI","benchmark":"DeepSWE v1.1","score":66.6,"unit":"%","setting":"Max effort","reportedBy":"Lab (self-reported)","publisher":"OpenAI","published":null,"source":"https://openai.com/index/introducing-gpt-6-sol-and-luna/","checked":"2026-10-02"}},{"slug":"gpt-6-sol-max-deepswe-v1-1-openai","name":"GPT-6 Sol (max) — DeepSWE v1.1 (OpenAI, self-reported)","path":null,"category":null,"updatedAt":"2026-10-02T12:09:10.242Z","fields":{"model":"GPT-6 Sol (max)","lab":"OpenAI","benchmark":"DeepSWE v1.1","score":68.8,"unit":"%","setting":"Max effort","reportedBy":"Lab (self-reported)","publisher":"OpenAI","published":null,"source":"https://openai.com/index/introducing-gpt-6-sol-and-luna/","checked":"2026-10-02"}},{"slug":"grok-4-7-high-deepswe-v1-1-xai","name":"Grok 4.7 (high) — DeepSWE v1.1 (xAI, self-reported)","path":null,"category":null,"updatedAt":"2026-10-05T05:11:17.391Z","fields":{"model":"Grok 4.7 (high)","lab":"xAI","benchmark":"DeepSWE v1.1","score":71,"unit":"%","setting":"High effort","reportedBy":"Lab (self-reported)","publisher":"xAI","published":"2026-09-21","source":"https://x.ai/news/grok-4-7","checked":"2026-10-05"}},{"slug":"kimi-k3-max-deepswe-v1-1-moonshot","name":"Kimi K3 (max) — DeepSWE v1.1 (Moonshot AI, mini-SWE-agent harness)","path":null,"category":null,"updatedAt":"2026-10-05T05:11:30.293Z","fields":{"model":"Kimi K3 (max)","lab":"Moonshot AI","benchmark":"DeepSWE v1.1","score":67.3,"unit":"%","setting":"Max reasoning effort; mini-SWE-agent harness; temperature 1.0, top-p 1.0","reportedBy":"Lab (self-reported)","publisher":"Moonshot AI","published":"2026-07-17","source":"https://www.kimi.ai/blog/kimi-k3","checked":"2026-10-05"}},{"slug":"gpt-6-astra-frontiermath-tier-4-openai","name":"GPT-6 Astra — FrontierMath Tier 4 (OpenAI, self-reported)","path":null,"category":null,"updatedAt":"2026-10-02T12:13:21.025Z","fields":{"model":"GPT-6 Astra","lab":"OpenAI","benchmark":"FrontierMath Tier 4","score":97.6,"unit":"%","setting":"FrontierMath Tier 4 v2; maximum across effort levels","reportedBy":"Lab (self-reported)","publisher":"OpenAI","published":"2026-09-22","source":"https://openai.com/index/gpt-6-astra/","checked":"2026-10-02"}},{"slug":"claude-opus-5-5-max-gdpval-aa-v2-1","name":"Claude Opus 5.5 (max) — GDPval-AA v2.1 (Artificial Analysis)","path":null,"category":null,"updatedAt":"2026-10-05T05:09:58.907Z","fields":{"model":"Claude Opus 5.5 (max)","lab":"Anthropic","benchmark":"GDPval-AA v2.1","score":1846,"unit":"Elo","setting":"Max effort (Artificial Analysis run)","reportedBy":"Independent evaluator","publisher":"Artificial Analysis","published":"2026-09-22","source":"https://artificialanalysis.ai/articles/claude-opus-5-5","checked":"2026-10-05"}},{"slug":"gpt-6-astra-gpqa-diamond-openai","name":"GPT-6 Astra — GPQA Diamond (OpenAI, self-reported)","path":null,"category":null,"updatedAt":"2026-10-02T12:13:25.248Z","fields":{"model":"GPT-6 Astra","lab":"OpenAI","benchmark":"GPQA Diamond","score":96,"unit":"%","setting":"Maximum across effort levels (research environment)","reportedBy":"Lab (self-reported)","publisher":"OpenAI","published":"2026-09-22","source":"https://openai.com/index/gpt-6-astra/","checked":"2026-10-02"}},{"slug":"kimi-k3-max-gpqa-diamond-moonshot","name":"Kimi K3 (max) — GPQA Diamond (Moonshot AI model card)","path":null,"category":null,"updatedAt":"2026-10-05T05:11:32.450Z","fields":{"model":"Kimi K3 (max)","lab":"Moonshot AI","benchmark":"GPQA Diamond","score":93.5,"unit":"%","setting":"Max reasoning effort; temperature 1.0","reportedBy":"Lab (self-reported)","publisher":"Moonshot AI","published":null,"source":"https://huggingface.co/moonshotai/Kimi-K3","checked":"2026-10-05"}},{"slug":"gemini-3-8-flash-hle-verified-google","name":"Gemini 3.8 Flash — HLE-Verified (Google, self-reported)","path":null,"category":null,"updatedAt":"2026-10-05T05:10:59.231Z","fields":{"model":"Gemini 3.8 Flash","lab":"Google DeepMind","benchmark":"HLE-Verified","score":54.9,"unit":"%","setting":"As reported in Google's announcement","reportedBy":"Lab (self-reported)","publisher":"Google","published":"2026-09-02","source":"https://blog.google/innovation-and-ai/models-and-research/gemini-models/3-8-flash-and-3-8-flash-cyber/","checked":"2026-10-05"}},{"slug":"claude-fable-5-1-humanitys-last-exam-no-tools-anthropic","name":"Claude Fable 5.1 — Humanity's Last Exam, no tools (Anthropic, self-reported)","path":null,"category":null,"updatedAt":"2026-10-05T05:09:50.151Z","fields":{"model":"Claude Fable 5.1","lab":"Anthropic","benchmark":"Humanity's Last Exam","score":60.9,"unit":"%","setting":"No tools","reportedBy":"Lab (self-reported)","publisher":"Anthropic","published":null,"source":"https://www.anthropic.com/claude-fable-and-mythos-5-1","checked":"2026-10-05"}},{"slug":"claude-opus-5-5-humanitys-last-exam-with-tools-anthropic","name":"Claude Opus 5.5 — Humanity's Last Exam, with tools (Anthropic, self-reported)","path":null,"category":null,"updatedAt":"2026-10-05T05:10:22.184Z","fields":{"model":"Claude Opus 5.5","lab":"Anthropic","benchmark":"Humanity's Last Exam","score":67.7,"unit":"%","setting":"With tools","reportedBy":"Lab (self-reported)","publisher":"Anthropic","published":"2026-09-22","source":"https://www.anthropic.com/claude-opus-5-5","checked":"2026-10-05"}},{"slug":"claude-opus-5-5-max-humanitys-last-exam-artificial-analysis","name":"Claude Opus 5.5 (max) — Humanity's Last Exam (Artificial Analysis)","path":null,"category":null,"updatedAt":"2026-10-05T05:10:01.800Z","fields":{"model":"Claude Opus 5.5 (max)","lab":"Anthropic","benchmark":"Humanity's Last Exam","score":61.4,"unit":"%","setting":"Max effort (Artificial Analysis run)","reportedBy":"Independent evaluator","publisher":"Artificial Analysis","published":"2026-09-22","source":"https://artificialanalysis.ai/articles/claude-opus-5-5","checked":"2026-10-05"}},{"slug":"gemini-4-argon-lvbench-google","name":"Gemini 4 Argon — LVBench (Google, self-reported)","path":null,"category":null,"updatedAt":"2026-10-05T05:11:11.313Z","fields":{"model":"Gemini 4 Argon","lab":"Google DeepMind","benchmark":"LVBench","score":91.7,"unit":"%","setting":"As reported in Google's announcement","reportedBy":"Lab (self-reported)","publisher":"Google","published":"2026-09-30","source":"https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-4-argon/","checked":"2026-10-05"}},{"slug":"gpt-6-astra-osworld-2-0-openai","name":"GPT-6 Astra — OSWorld 2.0 (OpenAI, self-reported)","path":null,"category":null,"updatedAt":"2026-10-02T12:06:54.540Z","fields":{"model":"GPT-6 Astra","lab":"OpenAI","benchmark":"OSWorld 2.0","score":72.6,"unit":"%","setting":"Maximum across effort levels","reportedBy":"Lab (self-reported)","publisher":"OpenAI","published":"2026-09-22","source":"https://openai.com/index/gpt-6-astra/","checked":"2026-10-02"}},{"slug":"claude-opus-5-5-osworld-2-1-anthropic","name":"Claude Opus 5.5 — OSWorld 2.1 (Anthropic, self-reported, partial)","path":null,"category":null,"updatedAt":"2026-10-05T05:10:24.000Z","fields":{"model":"Claude Opus 5.5","lab":"Anthropic","benchmark":"OSWorld 2.1","score":81.8,"unit":"%","setting":"Partial-credit scoring","reportedBy":"Lab (self-reported)","publisher":"Anthropic","published":"2026-09-22","source":"https://www.anthropic.com/claude-opus-5-5","checked":"2026-10-05"}},{"slug":"claude-sonnet-5-5-osworld-2-1-anthropic","name":"Claude Sonnet 5.5 — OSWorld 2.1 (Anthropic, self-reported, partial)","path":null,"category":null,"updatedAt":"2026-10-05T05:10:34.544Z","fields":{"model":"Claude Sonnet 5.5","lab":"Anthropic","benchmark":"OSWorld 2.1","score":80.1,"unit":"%","setting":"Partial-credit scoring","reportedBy":"Lab (self-reported)","publisher":"Anthropic","published":"2026-09-28","source":"https://www.anthropic.com/claude-sonnet-5-5","checked":"2026-10-05"}},{"slug":"claude-opus-5-5-max-scicode-artificial-analysis","name":"Claude Opus 5.5 (max) — SciCode (Artificial Analysis)","path":null,"category":null,"updatedAt":"2026-10-05T05:10:04.820Z","fields":{"model":"Claude Opus 5.5 (max)","lab":"Anthropic","benchmark":"SciCode","score":66.9,"unit":"%","setting":"Max effort (Artificial Analysis run)","reportedBy":"Independent evaluator","publisher":"Artificial Analysis","published":"2026-09-22","source":"https://artificialanalysis.ai/articles/claude-opus-5-5","checked":"2026-10-05"}},{"slug":"gpt-5-6-sol-terminal-bench-2-1-openai","name":"GPT-5.6 Sol — Terminal-Bench 2.1 (OpenAI, self-reported)","path":null,"category":null,"updatedAt":"2026-10-02T12:12:30.224Z","fields":{"model":"GPT-5.6 Sol","lab":"OpenAI","benchmark":"Terminal-Bench 2.1","score":88.8,"unit":"%","setting":"As reported in OpenAI's announcement","reportedBy":"Lab (self-reported)","publisher":"OpenAI","published":"2026-07-09","source":"https://openai.com/index/gpt-5-6/","checked":"2026-10-02"}},{"slug":"kimi-k3-max-terminal-bench-2-1-moonshot","name":"Kimi K3 (max) — Terminal-Bench 2.1 (Moonshot AI model card)","path":null,"category":null,"updatedAt":"2026-10-05T05:11:35.668Z","fields":{"model":"Kimi K3 (max)","lab":"Moonshot AI","benchmark":"Terminal-Bench 2.1","score":88.3,"unit":"%","setting":"Max reasoning effort; Kimi Code harness","reportedBy":"Lab (self-reported)","publisher":"Moonshot AI","published":null,"source":"https://huggingface.co/moonshotai/Kimi-K3","checked":"2026-10-05"}},{"slug":"claude-fable-5-1-terminal-bench-4-anthropic","name":"Claude Fable 5.1 — Terminal-Bench 4.0 (Anthropic, self-reported)","path":null,"category":null,"updatedAt":"2026-10-05T05:09:53.130Z","fields":{"model":"Claude Fable 5.1","lab":"Anthropic","benchmark":"Terminal-Bench 4.0","score":55.8,"unit":"%","setting":"As reported in Anthropic's table","reportedBy":"Lab (self-reported)","publisher":"Anthropic","published":null,"source":"https://www.anthropic.com/claude-fable-and-mythos-5-1","checked":"2026-10-05"}},{"slug":"claude-opus-5-5-max-terminal-bench-4-artificial-analysis","name":"Claude Opus 5.5 (max) — Terminal-Bench 4.0 (Artificial Analysis)","path":null,"category":null,"updatedAt":"2026-10-05T05:10:07.604Z","fields":{"model":"Claude Opus 5.5 (max)","lab":"Anthropic","benchmark":"Terminal-Bench 4.0","score":59.6,"unit":"%","setting":"Max effort (Artificial Analysis run)","reportedBy":"Independent evaluator","publisher":"Artificial Analysis","published":"2026-09-22","source":"https://artificialanalysis.ai/articles/claude-opus-5-5","checked":"2026-10-05"}},{"slug":"claude-opus-5-5-xhigh-terminal-bench-4-anthropic","name":"Claude Opus 5.5 (xhigh) — Terminal-Bench 4.0 (Anthropic, self-reported)","path":null,"category":null,"updatedAt":"2026-10-05T05:10:10.985Z","fields":{"model":"Claude Opus 5.5 (xhigh)","lab":"Anthropic","benchmark":"Terminal-Bench 4.0","score":66.4,"unit":"%","setting":"xhigh effort","reportedBy":"Lab (self-reported)","publisher":"Anthropic","published":"2026-09-22","source":"https://www.anthropic.com/claude-opus-5-5","checked":"2026-10-05"}},{"slug":"claude-sonnet-5-5-terminal-bench-4-anthropic","name":"Claude Sonnet 5.5 — Terminal-Bench 4.0 (Anthropic, self-reported)","path":null,"category":null,"updatedAt":"2026-10-05T05:10:37.389Z","fields":{"model":"Claude Sonnet 5.5","lab":"Anthropic","benchmark":"Terminal-Bench 4.0","score":70.6,"unit":"%","setting":"As reported in Anthropic's announcement","reportedBy":"Lab (self-reported)","publisher":"Anthropic","published":"2026-09-28","source":"https://www.anthropic.com/claude-sonnet-5-5","checked":"2026-10-05"}},{"slug":"claude-sonnet-5-5-max-terminal-bench-4-artificial-analysis","name":"Claude Sonnet 5.5 (max) — Terminal-Bench 4.0 (Artificial Analysis)","path":null,"category":null,"updatedAt":"2026-10-05T05:10:28.711Z","fields":{"model":"Claude Sonnet 5.5 (max)","lab":"Anthropic","benchmark":"Terminal-Bench 4.0","score":64,"unit":"%","setting":"Max effort (Artificial Analysis run)","reportedBy":"Independent evaluator","publisher":"Artificial Analysis","published":"2026-09-30","source":"https://artificialanalysis.ai/articles/gemini-4-argon-google-top-three-labs","checked":"2026-10-05"}},{"slug":"gemini-4-argon-terminal-bench-4-artificial-analysis","name":"Gemini 4 Argon — Terminal-Bench 4.0 (Artificial Analysis)","path":null,"category":null,"updatedAt":"2026-10-05T05:11:14.400Z","fields":{"model":"Gemini 4 Argon","lab":"Google DeepMind","benchmark":"Terminal-Bench 4.0","score":57,"unit":"%","setting":"Artificial Analysis run","reportedBy":"Independent evaluator","publisher":"Artificial Analysis","published":"2026-09-30","source":"https://artificialanalysis.ai/articles/gemini-4-argon-google-top-three-labs","checked":"2026-10-05"}},{"slug":"gpt-6-astra-terminal-bench-4-openai","name":"GPT-6 Astra — Terminal-Bench 4.0 (OpenAI, self-reported)","path":null,"category":null,"updatedAt":"2026-10-02T12:09:00.405Z","fields":{"model":"GPT-6 Astra","lab":"OpenAI","benchmark":"Terminal-Bench 4.0","score":57.9,"unit":"%","setting":"Maximum across effort levels","reportedBy":"Lab (self-reported)","publisher":"OpenAI","published":"2026-09-22","source":"https://openai.com/index/gpt-6-astra/","checked":"2026-10-02"}},{"slug":"grok-4-7-terminal-bench-4-xai","name":"Grok 4.7 — Terminal-Bench 4.0 (xAI, self-reported)","path":null,"category":null,"updatedAt":"2026-10-05T05:11:23.280Z","fields":{"model":"Grok 4.7","lab":"xAI","benchmark":"Terminal-Bench 4.0","score":37.6,"unit":"%","setting":"As reported in xAI's announcement","reportedBy":"Lab (self-reported)","publisher":"xAI","published":"2026-09-21","source":"https://x.ai/news/grok-4-7","checked":"2026-10-05"}}]}