{
  "schemaVersion": "1.0",
  "name": "BenchLM benchmark definitions",
  "description": "Benchmark metadata, scoring weights, and model-score coverage counts.",
  "canonicalUrl": "https://benchlm.ai/data/benchmarks.json",
  "generatedAt": "2026-09-02T00:50:21.010Z",
  "sourceLastUpdated": "September 1, 2026",
  "sourceFiles": [
    "src/data/benchmark_descriptions.json",
    "src/data/scoring.js"
  ],
  "counts": {
    "categories": 10,
    "benchmarks": 408
  },
  "items": [
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "healthBench",
      "name": "HealthBench (raw)",
      "fullName": "HealthBench raw score",
      "description": "Raw score on realistic multi-turn healthcare conversations graded against expert-written rubrics.",
      "paperUrl": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",
      "paperTitle": "Claude Opus 5 System Card",
      "authors": "Anthropic",
      "year": "2026",
      "tasks": "5,000 multi-turn patient conversations",
      "format": "Raw rubric score",
      "difficulty": "Realistic healthcare conversations",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/healthbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/healthbench.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "healthBenchLengthAdjusted",
      "name": "HealthBench (length-adjusted)",
      "fullName": "HealthBench length-adjusted score",
      "description": "HealthBench score after applying a verbosity penalty to model responses.",
      "paperUrl": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",
      "paperTitle": "Claude Opus 5 System Card",
      "authors": "Anthropic",
      "year": "2026",
      "tasks": "5,000 multi-turn patient conversations",
      "format": "Length-adjusted rubric score",
      "difficulty": "Realistic healthcare conversations",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/healthbenchlengthadjusted",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/healthbenchlengthadjusted.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "healthBenchProfessionalRaw",
      "name": "HealthBench Professional (raw)",
      "fullName": "HealthBench Professional raw score",
      "description": "Raw score on physician-authored clinical consult, documentation, and research conversations.",
      "paperUrl": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",
      "paperTitle": "Claude Opus 5 System Card",
      "authors": "Anthropic",
      "year": "2026",
      "tasks": "525 physician-authored conversations",
      "format": "Raw rubric score",
      "difficulty": "Professional clinical tasks",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/healthbenchprofessionalraw",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/healthbenchprofessionalraw.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "bioMysteryBenchHumanSolvable",
      "name": "BioMysteryBench (human-solvable)",
      "fullName": "BioMysteryBench Human Solvable",
      "description": "Computational biology challenges that independent human experts were able to solve.",
      "paperUrl": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",
      "paperTitle": "Claude Opus 5 System Card",
      "authors": "Anthropic",
      "year": "2026",
      "tasks": "Human-solvable computational biology investigations",
      "format": "Task score",
      "difficulty": "Expert computational biology",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/biomysterybenchhumansolvable",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/biomysterybenchhumansolvable.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "bioMysteryBenchHumanDifficult",
      "name": "BioMysteryBench (human-difficult)",
      "fullName": "BioMysteryBench Human Difficult",
      "description": "Computational biology challenges with objective answers that remained unsolved by independent human experts.",
      "paperUrl": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",
      "paperTitle": "Claude Opus 5 System Card",
      "authors": "Anthropic",
      "year": "2026",
      "tasks": "Human-difficult computational biology investigations",
      "format": "Task score",
      "difficulty": "Frontier computational biology",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/biomysterybenchhumandifficult",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/biomysterybenchhumandifficult.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "spatialBenchVerified",
      "name": "SpatialBench Verified",
      "fullName": "LatchBio SpatialBench Verified",
      "description": "Analysis of spatial transcriptomics data across externally validated biological problems.",
      "paperUrl": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",
      "paperTitle": "Claude Opus 5 System Card",
      "authors": "LatchBio and Anthropic",
      "year": "2026",
      "tasks": "115 externally validated spatial transcriptomics problems",
      "format": "Task score",
      "difficulty": "Professional bioinformatics",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/spatialbenchverified",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/spatialbenchverified.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "singleCellBench",
      "name": "SingleCellBench",
      "fullName": "LatchBio SingleCellBench",
      "description": "Single-cell RNA sequencing analysis tasks spanning common bioinformatics workflows.",
      "paperUrl": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",
      "paperTitle": "Claude Opus 5 System Card",
      "authors": "LatchBio and Anthropic",
      "year": "2026",
      "tasks": "195 single-cell RNA sequencing problems",
      "format": "Task score",
      "difficulty": "Professional bioinformatics",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/singlecellbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/singlecellbench.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "proteinGymHard",
      "name": "ProteinGym Hard",
      "fullName": "ProteinGym Hard",
      "description": "Predicts mutation effects by ranking mutant protein sequences against wild type and comparing against laboratory measurements.",
      "paperUrl": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",
      "paperTitle": "Claude Opus 5 System Card",
      "authors": "Anthropic",
      "year": "2026",
      "tasks": "Hard protein mutation-effect ranking tasks",
      "format": "Rank correlation",
      "difficulty": "Computational protein science",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/proteingymhard",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/proteingymhard.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "proteinDesign",
      "name": "Protein Design",
      "fullName": "Anthropic Protein Design evaluation",
      "description": "Generates novel protein sequences under family, topology, globularity, and structural-motif constraints.",
      "paperUrl": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",
      "paperTitle": "Claude Opus 5 System Card",
      "authors": "Anthropic",
      "year": "2026",
      "tasks": "Constrained protein-sequence design tasks",
      "format": "Composite score",
      "difficulty": "Computational protein design",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/proteindesign",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/proteindesign.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "organicChemistryV2",
      "name": "Organic chemistry V2",
      "fullName": "Anthropic Organic Chemistry V2 evaluation",
      "description": "Chemistry tasks covering spectroscopy, synthesis planning, reaction prediction, and chemical structure images.",
      "paperUrl": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",
      "paperTitle": "Claude Opus 5 System Card",
      "authors": "Anthropic",
      "year": "2026",
      "tasks": "Organic chemistry reasoning tasks",
      "format": "Task score",
      "difficulty": "Expert organic chemistry",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/organicchemistryv2",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/organicchemistryv2.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "protocolsTroubleshooting",
      "name": "Protocols (troubleshooting)",
      "fullName": "Molecular Biology Protocols Troubleshooting",
      "description": "Detects and fixes errors in molecular-biology protocols using document, code, and web-search tools.",
      "paperUrl": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",
      "paperTitle": "Claude Opus 5 System Card",
      "authors": "Anthropic",
      "year": "2026",
      "tasks": "Molecular-biology protocol troubleshooting",
      "format": "Task score",
      "difficulty": "Expert laboratory protocols",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/protocolstroubleshooting",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/protocolstroubleshooting.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "protocolsUnderstanding",
      "name": "Protocols (understanding)",
      "fullName": "Benchling Molecular Biology Protocols Understanding",
      "description": "Extends online molecular-biology protocols in additional directions using document and web-search tools.",
      "paperUrl": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",
      "paperTitle": "Claude Opus 5 System Card",
      "authors": "Benchling and Anthropic",
      "year": "2026",
      "tasks": "Molecular-biology protocol extension",
      "format": "Task score",
      "difficulty": "Expert laboratory protocols",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/protocolsunderstanding",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/protocolsunderstanding.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "aaOpennessIndex",
      "name": "AA Openness Index",
      "fullName": "Artificial Analysis Openness Index",
      "description": "A display-only Artificial Analysis model-openness index.",
      "paperUrl": "https://artificialanalysis.ai/evaluations/artificial-analysis-openness-index",
      "paperTitle": "Artificial Analysis Openness Index",
      "authors": "Artificial Analysis",
      "year": "2026",
      "tasks": "Model openness assessment",
      "format": "Index score",
      "difficulty": "Display-only external reference",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 11,
      "url": "https://benchlm.ai/benchmarks/aaopennessindex",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/aaopennessindex.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "aaMmluPro",
      "name": "AA MMLU-Pro",
      "fullName": "Artificial Analysis MMLU-Pro",
      "description": "An independently evaluated MMLU-Pro result from Artificial Analysis.",
      "paperUrl": "https://artificialanalysis.ai/evaluations/mmlu-pro",
      "paperTitle": "Artificial Analysis MMLU-Pro Benchmark Leaderboard",
      "authors": "Artificial Analysis",
      "year": "2026",
      "tasks": "Professional multi-subject questions",
      "format": "Accuracy",
      "difficulty": "Professional knowledge and reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 4,
      "url": "https://benchlm.ai/benchmarks/aammlupro",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/aammlupro.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "mmlu",
      "name": "MMLU",
      "fullName": "Massive Multitask Language Understanding",
      "description": "A comprehensive multiple-choice question answering test covering 57 tasks including elementary mathematics, US history, computer science, law, and more. Tests knowledge across diverse academic subjects from high school to professional level.",
      "paperUrl": "https://arxiv.org/abs/2009.03300",
      "paperTitle": "Measuring Massive Multitask Language Understanding",
      "authors": "Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt",
      "year": "2020",
      "tasks": "57 subjects",
      "format": "Multiple choice questions",
      "difficulty": "Elementary to professional level",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 6,
      "url": "https://benchlm.ai/benchmarks/mmlu",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mmlu.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "gpqa",
      "name": "GPQA",
      "fullName": "Graduate-Level Google-Proof Q&A",
      "description": "A challenging dataset of 448 multiple-choice questions written by domain experts in biology, physics, and chemistry. Designed to be difficult even for skilled non-experts with access to Google.",
      "paperUrl": "https://arxiv.org/abs/2311.12022",
      "paperTitle": "GPQA: A Graduate-Level Google-Proof Q&A Benchmark",
      "authors": "David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman",
      "year": "2023",
      "tasks": "448 questions",
      "format": "Multiple choice questions",
      "difficulty": "Graduate level",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": 0.07,
      "displayableScoreCount": 78,
      "url": "https://benchlm.ai/benchmarks/gpqa",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/gpqa.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "gpqaDiamond",
      "name": "GPQA-D",
      "fullName": "GPQA Diamond",
      "description": "A display-only GPQA Diamond reference from provider comparison charts.",
      "paperUrl": "https://www.arcee.ai/blog/trinity-large-thinking",
      "paperTitle": "Trinity-Large-Thinking: Scaling an Open Source Frontier Agent",
      "authors": "Arcee AI",
      "year": "2026",
      "tasks": "Graduate-level science questions",
      "format": "Multiple choice questions",
      "difficulty": "Graduate level",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 52,
      "url": "https://benchlm.ai/benchmarks/gpqa-diamond",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/gpqa-diamond.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "superGpqa",
      "name": "SuperGPQA",
      "fullName": "SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines",
      "description": "An expanded version of GPQA that evaluates graduate-level knowledge and reasoning capabilities across 285 disciplines, providing comprehensive coverage of academic domains.",
      "paperUrl": "https://arxiv.org/abs/2502.14739",
      "paperTitle": "SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines",
      "authors": "Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou",
      "year": "2025",
      "tasks": "285 disciplines",
      "format": "Multiple choice questions",
      "difficulty": "Graduate level",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": 0.07,
      "displayableScoreCount": 17,
      "url": "https://benchlm.ai/benchmarks/supergpqa",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/supergpqa.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "mmluPro",
      "name": "MMLU-Pro",
      "fullName": "Massive Multitask Language Understanding Professional",
      "description": "An enhanced version of MMLU with 10 answer choices instead of 4, featuring more reasoning-focused questions that better differentiate frontier models.",
      "paperUrl": "https://arxiv.org/abs/2406.01574",
      "paperTitle": "MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark",
      "authors": "Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen",
      "year": "2024",
      "tasks": "Multiple subjects",
      "format": "10-way multiple choice",
      "difficulty": "Professional level",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": 0.3,
      "displayableScoreCount": 39,
      "url": "https://benchlm.ai/benchmarks/mmlu-pro",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mmlu-pro.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "agieval",
      "name": "AGIEval",
      "fullName": "AGIEval",
      "description": "A human-centric exam benchmark for general knowledge and reasoning reported in DeepSeek-V4 base-model evaluations.",
      "paperUrl": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf",
      "paperTitle": "DeepSeek-V4 Technical Report",
      "authors": "DeepSeek-AI",
      "year": "2026",
      "tasks": "General academic and professional exam questions",
      "format": "Exact match",
      "difficulty": "General knowledge",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/agieval",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/agieval.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "hle",
      "name": "HLE",
      "fullName": "Humanity's Last Exam",
      "description": "An expert-authored benchmark designed to probe frontier knowledge and reasoning. BenchLM keeps protocol differences visible because tool-assisted and closed-book HLE runs answer different questions.",
      "paperUrl": "https://lastexam.ai/",
      "paperTitle": "Humanity's Last Exam",
      "authors": "Center for AI Safety, Scale AI, and thousands of expert contributors",
      "year": "2025",
      "tasks": "Expert-level questions",
      "format": "Open-ended and multiple choice",
      "difficulty": "Frontier expert level",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": 0.45,
      "displayableScoreCount": 53,
      "url": "https://benchlm.ai/benchmarks/hle",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/hle.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "frontierScience",
      "name": "FrontierScience",
      "fullName": "FrontierScience",
      "description": "A benchmark for research-level scientific reasoning, designed to separate frontier models on difficult science tasks that mix domain knowledge with deep reasoning.",
      "paperUrl": "https://openai.com/index/frontierscience/",
      "paperTitle": "FrontierScience",
      "authors": "OpenAI",
      "year": "2026",
      "tasks": "Research-level science tasks",
      "format": "Scientific reasoning benchmark",
      "difficulty": "Research frontier",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/frontierscience",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/frontierscience.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "artificialAnalysis",
      "name": "Artificial Analysis Intelligence Index",
      "fullName": "Artificial Analysis Intelligence Index",
      "description": "A display-only intelligence index published by Artificial Analysis that aggregates provider-reported and benchmark-derived signals into a single model-level score.",
      "paperUrl": "https://artificialanalysis.ai/",
      "paperTitle": "Artificial Analysis",
      "authors": "Artificial Analysis",
      "year": "2026",
      "tasks": "Cross-benchmark intelligence index",
      "format": "Aggregated model score",
      "difficulty": "Display-only external reference",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 169,
      "url": "https://benchlm.ai/benchmarks/artificialanalysis",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/artificialanalysis.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "aaGpqaDiamond",
      "name": "AA-GPQA Diamond",
      "fullName": "Artificial Analysis GPQA Diamond",
      "description": "A display-only Artificial Analysis GPQA Diamond score.",
      "paperUrl": "https://artificialanalysis.ai/evaluations/gpqa-diamond",
      "paperTitle": "Artificial Analysis GPQA Diamond Benchmark Leaderboard",
      "authors": "Artificial Analysis",
      "year": "2026",
      "tasks": "Graduate-level science questions",
      "format": "Accuracy",
      "difficulty": "Graduate-level science reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 166,
      "url": "https://benchlm.ai/benchmarks/aagpqadiamond",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/aagpqadiamond.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "aaHle",
      "name": "AA-HLE",
      "fullName": "Artificial Analysis Humanity's Last Exam",
      "description": "A display-only Artificial Analysis Humanity's Last Exam score.",
      "paperUrl": "https://artificialanalysis.ai/evaluations/hle",
      "paperTitle": "Artificial Analysis Humanity's Last Exam Benchmark Leaderboard",
      "authors": "Artificial Analysis",
      "year": "2026",
      "tasks": "Expert-level questions",
      "format": "Accuracy",
      "difficulty": "Frontier expert reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 165,
      "url": "https://benchlm.ai/benchmarks/aahle",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/aahle.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "aaOmniscienceIndex",
      "name": "AA-Omniscience Index",
      "fullName": "Artificial Analysis Omniscience Index",
      "description": "A display-only Artificial Analysis factual knowledge index.",
      "paperUrl": "https://artificialanalysis.ai/evaluations/omniscience",
      "paperTitle": "AA-Omniscience: Knowledge and Hallucination Benchmark",
      "authors": "Artificial Analysis",
      "year": "2026",
      "tasks": "Knowledge questions",
      "format": "Index score",
      "difficulty": "Broad factual knowledge",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 155,
      "url": "https://benchlm.ai/benchmarks/aaomniscienceindex",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/aaomniscienceindex.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "omniscienceAccuracy",
      "name": "AA-Omniscience Accuracy",
      "fullName": "Artificial Analysis Omniscience Accuracy",
      "description": "A display-only Artificial Analysis knowledge metric for the proportion of correctly answered questions.",
      "paperUrl": "https://artificialanalysis.ai/models/grok-4-3",
      "paperTitle": "Artificial Analysis model benchmarks",
      "authors": "Artificial Analysis",
      "year": "2026",
      "tasks": "Knowledge questions",
      "format": "Accuracy",
      "difficulty": "Broad knowledge",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 154,
      "url": "https://benchlm.ai/benchmarks/omniscienceaccuracy",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/omniscienceaccuracy.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "omniscienceHallucinationRate",
      "name": "AA-Omniscience Hallucination Rate",
      "fullName": "Artificial Analysis Omniscience Hallucination Rate",
      "description": "A display-only Artificial Analysis factuality metric for the rate of incorrect answers among non-correct responses.",
      "paperUrl": "https://artificialanalysis.ai/models/grok-4-3",
      "paperTitle": "Artificial Analysis model benchmarks",
      "authors": "Artificial Analysis",
      "year": "2026",
      "tasks": "Knowledge questions",
      "format": "Hallucination rate",
      "difficulty": "Factuality",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 154,
      "url": "https://benchlm.ai/benchmarks/omnisciencehallucinationrate",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/omnisciencehallucinationrate.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "simpleQa",
      "name": "SimpleQA",
      "fullName": "Measuring Short-Form Factuality in Large Language Models",
      "description": "A benchmark that evaluates the ability of language models to answer short, fact-seeking questions accurately. Focuses on factual correctness rather than reasoning complexity.",
      "paperUrl": "https://arxiv.org/abs/2411.04368",
      "paperTitle": "Measuring short-form factuality in large language models",
      "authors": "Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer",
      "year": "2024",
      "tasks": "Factual questions",
      "format": "Short-form Q&A",
      "difficulty": "Factual accuracy focused",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": 0.11,
      "displayableScoreCount": 3,
      "url": "https://benchlm.ai/benchmarks/simpleqa",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/simpleqa.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "chineseSimpleQa",
      "name": "Chinese-SimpleQA",
      "fullName": "Chinese-SimpleQA",
      "description": "A Chinese short-form factuality benchmark reported by DeepSeek for V4 model evaluations.",
      "paperUrl": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf",
      "paperTitle": "DeepSeek-V4 Technical Report",
      "authors": "DeepSeek-AI",
      "year": "2026",
      "tasks": "Chinese factual questions",
      "format": "Short-form factual QA",
      "difficulty": "Factual accuracy focused",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/chinesesimpleqa",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/chinesesimpleqa.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "openBookQa",
      "name": "OpenBookQA",
      "fullName": "OpenBookQA",
      "description": "A science question-answering benchmark that tests whether models can apply a small open-book set of elementary science facts to multi-step reasoning questions.",
      "paperUrl": "https://arxiv.org/abs/1809.02789",
      "paperTitle": "Can a Suit of Armor Conduct Electricity? A New Dataset for Open Book Question Answering",
      "authors": "Todor Mihaylov, Peter Clark, Tushar Khot, Ashish Sabharwal",
      "year": "2018",
      "tasks": "Elementary science questions",
      "format": "4-way multiple choice",
      "difficulty": "Elementary science reasoning",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/openbookqa",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/openbookqa.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "healthBenchHard",
      "name": "HealthBench Hard",
      "fullName": "HealthBench Hard",
      "description": "A harder subset of OpenAI's HealthBench for evaluating open-ended medical and health reasoning with rubric-based grading.",
      "paperUrl": "https://ai.meta.com/static-resource/muse-spark-eval-methodology",
      "paperTitle": "Muse Spark Eval Methodology",
      "authors": "Meta AI",
      "year": "2026",
      "tasks": "1,000 health prompts",
      "format": "Open-ended health evaluation",
      "difficulty": "Advanced health reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 8,
      "url": "https://benchlm.ai/benchmarks/healthbench-hard",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/healthbench-hard.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "healthBenchProfessional",
      "name": "HealthBench Professional",
      "fullName": "HealthBench Professional",
      "description": "An open benchmark for clinician-facing model responses across care consult, writing and documentation, and medical research tasks.",
      "paperUrl": "https://arxiv.org/abs/2604.27470",
      "paperTitle": "HealthBench Professional: Evaluating Large Language Models on Real Clinician Chats",
      "authors": "Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal",
      "year": "2026",
      "tasks": "Clinician chat tasks",
      "format": "Rubric-graded open-ended responses",
      "difficulty": "Professional clinical workflows",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 6,
      "url": "https://benchlm.ai/benchmarks/healthbenchprofessional",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/healthbenchprofessional.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "medXpertQaText",
      "name": "MedXpertQA (Text)",
      "fullName": "MedXpertQA Text",
      "description": "A medical multiple-choice benchmark spanning many specialties with 10 answer options per question.",
      "paperUrl": "https://ai.meta.com/static-resource/muse-spark-eval-methodology",
      "paperTitle": "Muse Spark Eval Methodology",
      "authors": "Meta AI",
      "year": "2026",
      "tasks": "2,450 medical multiple-choice questions",
      "format": "Medical MCQ",
      "difficulty": "Professional medical knowledge",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 5,
      "url": "https://benchlm.ai/benchmarks/medxpertqatext",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/medxpertqatext.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "frontierScienceResearch",
      "name": "FrontierScience Research",
      "fullName": "FrontierScience Research",
      "description": "A research-focused FrontierScience evaluation variant for scientific investigation and problem solving.",
      "paperUrl": "https://ai.meta.com/static-resource/muse-spark-eval-methodology",
      "paperTitle": "Muse Spark Eval Methodology",
      "authors": "Meta AI",
      "year": "2026",
      "tasks": "Scientific research problems",
      "format": "Research evaluation",
      "difficulty": "Frontier scientific research",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 3,
      "url": "https://benchlm.ai/benchmarks/frontierscienceresearch",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/frontierscienceresearch.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "truthfulqa",
      "name": "TruthfulQA",
      "fullName": "TruthfulQA",
      "description": "A benchmark designed to measure whether language models produce truthful answers instead of repeating common misconceptions or misleading falsehoods.",
      "paperUrl": "https://arxiv.org/abs/2109.07958",
      "paperTitle": "TruthfulQA: Measuring How Models Mimic Human Falsehoods",
      "authors": "Stephanie Lin, Jacob Hilton, Owain Evans",
      "year": "2021",
      "tasks": "Truthfulness and misconception resistance",
      "format": "Question answering",
      "difficulty": "Hallucination and factuality stress test",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/truthfulqa",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/truthfulqa.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "hleNoTools",
      "name": "HLE w/o tools",
      "fullName": "Humanity's Last Exam without tools",
      "description": "Tool-free variant of Humanity's Last Exam that isolates a model's raw frontier reasoning.",
      "paperUrl": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
      "paperTitle": "Introducing GPT-5.4 mini and nano",
      "authors": "OpenAI",
      "year": "2026",
      "tasks": "Expert-level questions",
      "format": "Tool-free expert QA",
      "difficulty": "Frontier expert level",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 36,
      "url": "https://benchlm.ai/benchmarks/hlenotools",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/hlenotools.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "mmluProArcee",
      "name": "MMLU-Pro (Arcee)",
      "fullName": "MMLU-Pro first-party comparison snapshot",
      "description": "A display-only MMLU-Pro reference from Arcee AI's Trinity-Large-Thinking launch chart.",
      "paperUrl": "https://www.arcee.ai/blog/trinity-large-thinking",
      "paperTitle": "Trinity-Large-Thinking: Scaling an Open Source Frontier Agent",
      "authors": "Arcee AI",
      "year": "2026",
      "tasks": "Professional academic QA",
      "format": "10-way multiple choice",
      "difficulty": "Professional level",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 6,
      "url": "https://benchlm.ai/benchmarks/mmluproarcee",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mmluproarcee.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "mmluRedux",
      "name": "MMLU-Redux",
      "fullName": "MMLU-Redux",
      "description": "A harder refresh of MMLU intended to keep broad knowledge evaluation useful after the original benchmark became too easy for frontier models.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2026",
      "tasks": "Broad academic QA",
      "format": "Multiple choice questions",
      "difficulty": "Advanced general knowledge",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 9,
      "url": "https://benchlm.ai/benchmarks/mmluredux",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mmluredux.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "mmmlu",
      "name": "MMMLU",
      "fullName": "MMMLU",
      "description": "A multilingual MMLU-style benchmark reported in provider evaluation tables.",
      "paperUrl": "https://huggingface.co/datasets/openai/MMMLU",
      "paperTitle": "MMMLU",
      "authors": "OpenAI",
      "year": "2026",
      "tasks": "Multilingual academic QA",
      "format": "Exact match",
      "difficulty": "Broad multilingual knowledge",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 4,
      "url": "https://benchlm.ai/benchmarks/mmmlu",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mmmlu.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "cEval",
      "name": "C-Eval",
      "fullName": "C-Eval",
      "description": "A Chinese-language academic and professional benchmark spanning humanities, social science, STEM, and applied subjects.",
      "paperUrl": "https://arxiv.org/abs/2305.08322",
      "paperTitle": "C-Eval: A Multi-Level Multi-Discipline Chinese Evaluation Suite for Foundation Models",
      "authors": "C-Eval authors",
      "year": "2023",
      "tasks": "Chinese academic and professional exams",
      "format": "Multiple choice questions",
      "difficulty": "High school to professional level",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 5,
      "url": "https://benchlm.ai/benchmarks/ceval",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/ceval.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "cmmlu",
      "name": "CMMLU",
      "fullName": "Chinese Massive Multitask Language Understanding",
      "description": "A Chinese multitask academic benchmark reported in DeepSeek-V4 base-model evaluations.",
      "paperUrl": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf",
      "paperTitle": "DeepSeek-V4 Technical Report",
      "authors": "DeepSeek-AI",
      "year": "2026",
      "tasks": "Chinese academic QA",
      "format": "Exact match",
      "difficulty": "Broad Chinese knowledge",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/cmmlu",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/cmmlu.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "multiLoKo",
      "name": "MultiLoKo",
      "fullName": "MultiLoKo",
      "description": "A multilingual/localized knowledge benchmark reported in DeepSeek-V4 base-model evaluations.",
      "paperUrl": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf",
      "paperTitle": "DeepSeek-V4 Technical Report",
      "authors": "DeepSeek-AI",
      "year": "2026",
      "tasks": "Localized multilingual knowledge questions",
      "format": "Exact match",
      "difficulty": "Multilingual knowledge",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/multiloko",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/multiloko.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "factsParametric",
      "name": "FACTS Parametric",
      "fullName": "FACTS Parametric",
      "description": "A parametric factuality benchmark reported in DeepSeek-V4 base-model evaluations.",
      "paperUrl": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf",
      "paperTitle": "DeepSeek-V4 Technical Report",
      "authors": "DeepSeek-AI",
      "year": "2026",
      "tasks": "Parametric factual recall",
      "format": "Exact match",
      "difficulty": "Factual accuracy focused",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/factsparametric",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/factsparametric.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "triviaQa",
      "name": "TriviaQA",
      "fullName": "TriviaQA",
      "description": "A reading and trivia question-answering benchmark reported in DeepSeek-V4 base-model evaluations.",
      "paperUrl": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf",
      "paperTitle": "DeepSeek-V4 Technical Report",
      "authors": "DeepSeek-AI",
      "year": "2026",
      "tasks": "Trivia and reading-comprehension QA",
      "format": "Exact match",
      "difficulty": "General factual QA",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/triviaqa",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/triviaqa.md"
    },
    {
      "category": "knowledge",
      "categoryLabel": "Knowledge",
      "benchmarkKey": "financeArena",
      "name": "FinanceArena",
      "fullName": "FinanceArena — FinanceQA Assumption-Based",
      "description": "An AfterQuery benchmark of open-ended financial analysis that requires models to read financial data, make assumptions, and return exact answers.",
      "paperUrl": "https://arxiv.org/abs/2501.18062",
      "paperTitle": "FinanceQA",
      "authors": "AfterQuery",
      "year": "2025",
      "tasks": "Professional financial-analysis questions",
      "format": "Open-ended financial QA with exact-match grading",
      "difficulty": "Professional finance reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/financearena",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/financearena.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "programBenchEpisode1",
      "name": "ProgramBench (episode 1)",
      "fullName": "ProgramBench hidden-test pass rate after episode 1",
      "description": "Program-reconstruction hidden-test pass rate after the first of five sequential long-context episodes.",
      "paperUrl": "https://arxiv.org/abs/2605.03546",
      "paperTitle": "ProgramBench: Can language models rebuild programs from scratch?",
      "authors": "Yang et al.",
      "year": "2026",
      "tasks": "166 golden program-reconstruction tasks",
      "format": "Hidden-test pass rate after episode 1",
      "difficulty": "Long-context clean-room software engineering",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/programbenchepisode1",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/programbenchepisode1.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "aaLiveCodeBench",
      "name": "AA LiveCodeBench",
      "fullName": "Artificial Analysis LiveCodeBench",
      "description": "An independently evaluated LiveCodeBench result from Artificial Analysis.",
      "paperUrl": "https://artificialanalysis.ai/evaluations/livecodebench",
      "paperTitle": "Artificial Analysis LiveCodeBench Benchmark Leaderboard",
      "authors": "Artificial Analysis",
      "year": "2026",
      "tasks": "Contamination-resistant coding tasks",
      "format": "Pass rate",
      "difficulty": "Competitive programming",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 3,
      "url": "https://benchlm.ai/benchmarks/aalivecodebench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/aalivecodebench.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "aaTerminalBench21",
      "name": "AA Terminal-Bench 2.1",
      "fullName": "Artificial Analysis Terminal-Bench v2.1",
      "description": "An independently evaluated Terminal-Bench v2.1 result from Artificial Analysis.",
      "paperUrl": "https://artificialanalysis.ai/evaluations/terminalbench-v2-1",
      "paperTitle": "Artificial Analysis Terminal-Bench v2.1 Benchmark Leaderboard",
      "authors": "Artificial Analysis",
      "year": "2026",
      "tasks": "Terminal-based agent tasks",
      "format": "Task success rate",
      "difficulty": "Agentic software engineering",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": "https://benchlm.ai/benchmarks/terminal-bench-4",
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/aaterminalbench21",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/aaterminalbench21.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "terminalBench21",
      "name": "Terminal-Bench 2.1",
      "fullName": "Terminal-Bench 2.1 (provider run)",
      "description": "A provider-run Terminal-Bench 2.1 result stored separately from the repository's Terminal-Bench 2.0 lane.",
      "paperUrl": "https://api-docs.deepseek.com/zh-cn/updates/",
      "paperTitle": "DeepSeek V4 Flash 0731 update",
      "authors": "DeepSeek-AI",
      "year": "2026",
      "tasks": "Terminal-based software-agent tasks",
      "format": "Interactive task success rate",
      "difficulty": "Professional software engineering",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": "https://benchlm.ai/benchmarks/terminal-bench-4",
      "weight": null,
      "displayableScoreCount": 17,
      "url": "https://benchlm.ai/benchmarks/terminalbench21",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/terminalbench21.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "humaneval",
      "name": "HumanEval",
      "fullName": "Evaluating Large Language Models Trained on Code",
      "description": "A set of 164 handwritten Python function-generation problems. HumanEval is useful as a historical floor check, but BenchLM's current exact-source table is too small to support a broad frontier-coding verdict.",
      "paperUrl": "https://arxiv.org/abs/2107.03374",
      "paperTitle": "Evaluating Large Language Models Trained on Code",
      "authors": "Mark Chen, Jerry Tworek, Heewoo Jun, Qiming Yuan, Henrique Ponde de Oliveira Pinto, Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, Alex Ray, Raul Puri, Gretchen Krueger, Michael Petrov, Heidy Khlaaf, Girish Sastry, Pamela Mishkin, Brooke Chan, Scott Gray, Nick Ryder, Mikhail Pavlov, Alethea Power, Lukasz Kaiser, Mohammad Bavarian, Clemens Winter, Philippe Tillet, Felipe Petroski Such, Dave Cummings, Matthias Plappert, Fotios Chantzis, Elizabeth Barnes, Ariel Herbert-Voss, William Hebgen Guss, Alex Nichol, Alex Paino, Nikolas Tezak, Jie Tang, Igor Babuschkin, Suchir Balaji, Shantanu Jain, William Saunders, Christopher Hesse, Andrew N. Carr, Jan Leike, Josh Achiam, Vedant Misra, Evan Morikawa, Alec Radford, Matthew Knight, Miles Brundage, Mira Murati, Katie Mayer, Peter Welinder, Bob McGrew, Dario Amodei, Sam McCandlish, Ilya Sutskever, Wojciech Zaremba",
      "year": "2021",
      "tasks": "164 problems",
      "format": "Python function generation",
      "difficulty": "Introductory to intermediate programming",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/humaneval",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/humaneval.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "bigCodeBench",
      "name": "BigCodeBench",
      "fullName": "BigCodeBench",
      "description": "A code-generation benchmark reported in DeepSeek-V4 base-model evaluations.",
      "paperUrl": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf",
      "paperTitle": "DeepSeek-V4 Technical Report",
      "authors": "DeepSeek-AI",
      "year": "2026",
      "tasks": "Code generation tasks",
      "format": "Pass@1",
      "difficulty": "Software engineering",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/bigcodebench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/bigcodebench.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "codeforces",
      "name": "Codeforces",
      "fullName": "Codeforces Rating",
      "description": "Competitive-programming rating reported for DeepSeek-V4 thinking-mode evaluations.",
      "paperUrl": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf",
      "paperTitle": "DeepSeek-V4 Technical Report",
      "authors": "DeepSeek-AI",
      "year": "2026",
      "tasks": "Competitive programming contests",
      "format": "Rating",
      "difficulty": "Elite competitive programming",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 3,
      "url": "https://benchlm.ai/benchmarks/codeforces",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/codeforces.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "terminalBench2",
      "name": "Terminal-Bench 2.0",
      "fullName": "Terminal-Bench 2.0",
      "description": "A benchmark for agentic software engineering tasks executed in real terminal environments. DeepSeek reports it in the agentic section, while BenchLM also mirrors it in coding for models that publish it as a developer-task signal.",
      "paperUrl": "https://www.tbench.ai/",
      "paperTitle": "Terminal-Bench 2.0",
      "authors": "Terminal-Bench contributors",
      "year": "2026",
      "tasks": "Terminal-based software tasks",
      "format": "Interactive CLI agent evaluation",
      "difficulty": "Professional software engineering",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": "https://benchlm.ai/benchmarks/terminal-bench-4",
      "weight": null,
      "displayableScoreCount": 43,
      "url": "https://benchlm.ai/benchmarks/terminal-bench-2",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/terminal-bench-2.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "sweVerified",
      "name": "SWE-bench Verified",
      "fullName": "Software Engineering Benchmark Verified",
      "description": "A curated, human-verified subset of SWE-bench that tests models on resolving real GitHub issues from popular open-source Python repositories like Django, Flask, and scikit-learn.",
      "paperUrl": "https://arxiv.org/abs/2310.06770",
      "paperTitle": "SWE-bench: Can Language Models Resolve Real-World GitHub Issues?",
      "authors": "Carlos E. Jimenez, John Yang, Alexander Wettig, Shunyu Yao, Kexin Pei, Ofir Press, Karthik Narasimhan",
      "year": "2024",
      "tasks": "500 verified issues",
      "format": "Code patch generation",
      "difficulty": "Professional software engineering",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": 0.16,
      "displayableScoreCount": 65,
      "url": "https://benchlm.ai/benchmarks/swe-bench-verified",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/swe-bench-verified.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "sweRebench",
      "name": "SWE-Rebench",
      "fullName": "SWE-Rebench",
      "description": "A continuously updated software engineering benchmark by Nebius using fresh GitHub issues to avoid contamination. Models are evaluated 5 times per problem under a fixed ReAct scaffolding; the Resolved Rate (best pass@1) is reported.",
      "paperUrl": "https://swe-rebench.com",
      "paperTitle": "SWE-Rebench: Contamination-Free Evaluation of Software Engineering Agents",
      "authors": "Nebius",
      "year": "2026",
      "tasks": "Fresh GitHub issues (rolling window)",
      "format": "Code patch generation",
      "difficulty": "Professional software engineering",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": 0.2,
      "displayableScoreCount": 13,
      "url": "https://benchlm.ai/benchmarks/swe-rebench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/swe-rebench.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "liveCodeBench",
      "name": "LiveCodeBench",
      "fullName": "LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code",
      "description": "A continuously updated coding benchmark built from newly collected LeetCode, AtCoder, and Codeforces problems. Fresh problem windows reduce one contamination path, but results still need a release and setup check.",
      "paperUrl": "https://arxiv.org/abs/2403.07974",
      "paperTitle": "LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code",
      "authors": "Naman Jain, King Han, Alex Gu, Wen-Ding Li, Fanjia Yan, Tianjun Zhang, Sida Wang, Armando Solar-Lezama, Koushik Sen, Ion Stoica",
      "year": "2024",
      "tasks": "Continuously updated contest problems",
      "format": "Competitive-programming evaluation",
      "difficulty": "Competitive programming level",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": 0.38,
      "displayableScoreCount": 6,
      "url": "https://benchlm.ai/benchmarks/livecodebench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/livecodebench.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "liveCodeBenchV6",
      "name": "LiveCodeBench v6",
      "fullName": "LiveCodeBench v6",
      "description": "LiveCodeBench v6 is a named release slice used in provider comparison tables. Keeping it separate prevents v6 results from being mixed into older or rolling LiveCodeBench windows.",
      "paperUrl": "https://github.com/LiveCodeBench/LiveCodeBench",
      "paperTitle": "LiveCodeBench official repository and release documentation",
      "authors": "LiveCodeBench maintainers",
      "year": "2026",
      "tasks": "Fresh programming problems",
      "format": "Provider-published v6 competitive programming results",
      "difficulty": "Competitive programming level",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 21,
      "url": "https://benchlm.ai/benchmarks/livecodebench-v6",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/livecodebench-v6.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "liveCodeBenchV5",
      "name": "LiveCodeBench v5",
      "fullName": "LiveCodeBench v5",
      "description": "LiveCodeBench v5 is a named release and date-window slice. BenchLM keeps explicitly labeled v5 rows outside the rolling weighted lane.",
      "paperUrl": "https://github.com/LiveCodeBench/LiveCodeBench",
      "paperTitle": "LiveCodeBench official repository and release documentation",
      "authors": "LiveCodeBench maintainers",
      "year": "2025",
      "tasks": "July 2024 to May 2025 release window",
      "format": "Provider-published v5 competitive programming result",
      "difficulty": "Competitive programming level",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/livecodebench-v5",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/livecodebench-v5.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "liveCodeBenchPass1Cot",
      "name": "LiveCodeBench Pass@1-COT",
      "fullName": "LiveCodeBench Pass@1 with Chain-of-Thought",
      "description": "This lane contains DeepSeek's LiveCodeBench Pass@1-COT results. The explicit metric and prompting label keeps them separate from generic and version-specific LiveCodeBench rows.",
      "paperUrl": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/resolve/main/DeepSeek_V4.pdf?download=true",
      "paperTitle": "DeepSeek-V4 technical report",
      "authors": "DeepSeek",
      "year": "2026",
      "tasks": "DeepSeek-V4 report evaluation window",
      "format": "Pass@1-COT competitive programming results",
      "difficulty": "Competitive programming level",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/livecodebenchpass1cot",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/livecodebenchpass1cot.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "liveCodeBenchPro",
      "name": "LiveCodeBench Pro",
      "fullName": "LiveCodeBench Pro",
      "description": "A harder competitive-programming benchmark family built from Codeforces, ICPC, and IOI problems, with quarter-specific public leaderboards and difficulty-aware reporting.",
      "paperUrl": "https://arxiv.org/abs/2506.11928",
      "paperTitle": "LiveCodeBench Pro: How Do Olympiad Medalists Judge LLMs in Competitive Programming?",
      "authors": "LiveCodeBench Pro authors",
      "year": "2025",
      "tasks": "Quarter-specific contest programming sets",
      "format": "Competitive programming",
      "difficulty": "High-end contest programming",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 8,
      "url": "https://benchlm.ai/benchmarks/livecodebench-pro",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/livecodebench-pro.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "flteval",
      "name": "FLTEval",
      "fullName": "FLTEval",
      "description": "A repository-level Lean 4 proof engineering benchmark that measures whether a model can complete formal proofs and correctly define new mathematical concepts inside realistic FLT project pull requests.",
      "paperUrl": "https://mistral.ai/news/leanstral",
      "paperTitle": "Leanstral: Open-Source foundation for trustworthy vibe-coding",
      "authors": "Mistral AI",
      "year": "2026",
      "tasks": "FLT project pull requests",
      "format": "Lean 4 repository task completion",
      "difficulty": "Formal verification / proof engineering",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/flteval",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/flteval.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "swePro",
      "name": "SWE-bench Pro",
      "fullName": "SWE-bench Pro",
      "description": "A long-horizon repository benchmark built to test realistic software engineering work. Its scores need a task-quality and setup check before they support a coding-agent decision.",
      "paperUrl": "https://arxiv.org/abs/2509.16941",
      "paperTitle": "SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?",
      "authors": "Xiang Deng, Jeff Da, Edwin Pan, Yannis Yiming He, Charles Ide, Kanak Garg, Niklas Lauffer, Andrew Park, Nitin Pasari, Chetan Rane, Karmini Sampath, Maya Krishnan, Srivatsa Kundurthy, Sean Hendryx, Zifan Wang, Chen Bo Calvin Zhang, Noah Jacobson, Bing Liu, Brad Kenstler",
      "year": "2025",
      "tasks": "1,865 repository problems",
      "format": "Repository task completion",
      "difficulty": "Long-horizon professional engineering",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": 0.1,
      "displayableScoreCount": 63,
      "url": "https://benchlm.ai/benchmarks/swe-bench-pro",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/swe-bench-pro.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "seniorSweBench",
      "name": "Senior SWE-Bench",
      "fullName": "Senior SWE-Bench",
      "description": "A Snorkel AI benchmark of senior-level software engineering tasks emphasizing under-specified feature work, bug/performance investigation, and taste-based correctness.",
      "paperUrl": "https://senior-swe-bench.snorkel.ai/",
      "paperTitle": "Senior SWE-Bench",
      "authors": "Snorkel AI",
      "year": "2026",
      "tasks": "Senior-level repository tasks",
      "format": "Agentic software-engineering evaluation",
      "difficulty": "Professional senior engineering",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/seniorswebench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/seniorswebench.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "vulcanBench",
      "name": "VulcanBench v3",
      "fullName": "VulcanBench v3",
      "description": "An open software-engineering benchmark built from real merged post-cutoff pull requests across Python, Rust, TypeScript, JavaScript, and Go repositories.",
      "paperUrl": "https://github.com/morganlinton/VulcanBench/tree/main",
      "paperTitle": "VulcanBench",
      "authors": "VulcanBench contributors",
      "year": "2026",
      "tasks": "23 post-cutoff repository tasks in the v3 report",
      "format": "Pass@1 with low, medium, and high effort",
      "difficulty": "Professional multi-file software engineering",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 13,
      "url": "https://benchlm.ai/benchmarks/vulcanbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/vulcanbench.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "openHarmonyBench",
      "name": "OpenHarmony Bench",
      "fullName": "OpenHarmony Bench v1.0",
      "description": "An app-level coding benchmark that asks DevEco Code configurations to implement observable behavior in buildable OpenHarmony ArkTS applications.",
      "paperUrl": "https://arxiv.org/abs/2608.16022",
      "paperTitle": "OpenHarmony Bench: Evaluating LLMs and Coding Agents on OpenHarmony App Development",
      "authors": "OpenHarmony Bench authors",
      "year": "2026",
      "tasks": "153 app-development and bug-fix tasks",
      "format": "Task completion through DevEco Code",
      "difficulty": "End-to-end OpenHarmony application development",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 10,
      "url": "https://benchlm.ai/benchmarks/openharmonybench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/openharmonybench.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "vulcanCiiV1",
      "name": "VulcanBench CII v1",
      "fullName": "VulcanBench Coding Intelligence Index v1",
      "description": "A post-cutoff software-engineering benchmark with hidden functional tests and regression guards, reported for vendor coding-agent harnesses.",
      "paperUrl": "https://github.com/morganlinton/VulcanBench/blob/main/docs/results/cii-v1-2026-08/README.md",
      "paperTitle": "CII v1 frontier results",
      "authors": "VulcanBench contributors",
      "year": "2026",
      "tasks": "38 validated post-cutoff repository tasks",
      "format": "Pass@1 with vendor coding-agent harnesses",
      "difficulty": "Mid-band frontier software engineering",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/vulcanciiv1",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/vulcanciiv1.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "frontierCode",
      "name": "FrontierCode 1.1 Main",
      "fullName": "FrontierCode 1.1 Main",
      "description": "Cognition's 100-task software-engineering benchmark for whether coding agents produce mergeable, production-quality pull requests, scored for correctness, tests, scope, style, and maintainability through maintainer-authored rubrics.",
      "paperUrl": "https://cognition.com/frontiercode",
      "paperTitle": "FrontierCode leaderboard",
      "authors": "Cognition",
      "year": "2026",
      "tasks": "100 private Main tasks (150 in Extended)",
      "format": "Repository task completion with maintainer rubrics",
      "difficulty": "Frontier coding-agent quality",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 10,
      "url": "https://benchlm.ai/benchmarks/frontiercode",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/frontiercode.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "frontierCode11Extended",
      "name": "FrontierCode 1.1 Extended",
      "fullName": "FrontierCode 1.1 Extended",
      "description": "Cognition's 150-task Extended subset of the FrontierCode 1.1 software-engineering benchmark.",
      "paperUrl": "https://devin.ai/blog/gpt-5-6",
      "paperTitle": "GPT-5.6 models are now available in Devin",
      "authors": "Cognition",
      "year": "2026",
      "tasks": "150 private software-engineering tasks",
      "format": "Repository task completion with maintainer rubrics",
      "difficulty": "Frontier coding-agent quality",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 5,
      "url": "https://benchlm.ai/benchmarks/frontiercode11extended",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/frontiercode11extended.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "ideBench",
      "name": "IDE-Bench",
      "fullName": "IDE-Bench",
      "description": "An 80-task software-engineering benchmark across eight repositories that tests whether autonomous IDE agents can explore, edit, run, and verify code changes end to end.",
      "paperUrl": "https://arxiv.org/abs/2601.20886",
      "paperTitle": "IDE-Bench",
      "authors": "AfterQuery",
      "year": "2026",
      "tasks": "80 tasks across 8 repositories",
      "format": "Autonomous IDE-agent task completion (pass@1)",
      "difficulty": "End-to-end software engineering",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/idebench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/idebench.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "appBench",
      "name": "App-Bench",
      "fullName": "App-Bench",
      "description": "A six-task full-stack web-app benchmark that measures how much required functionality an AI builder or coding assistant delivers from one prompt without human code edits.",
      "paperUrl": "https://appbench.ai/",
      "paperTitle": "App-Bench",
      "authors": "AfterQuery",
      "year": "2025",
      "tasks": "6 full-stack app-building tasks",
      "format": "Best-of-three one-shot feature completion",
      "difficulty": "Production-style full-stack application generation",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/appbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/appbench.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "sweMultilingual",
      "name": "SWE Multilingual",
      "fullName": "SWE Multilingual",
      "description": "A multilingual software-engineering benchmark for real-world code issue resolution across multiple programming languages.",
      "paperUrl": "https://www.minimax.io/news/minimax-m27-en",
      "paperTitle": "MiniMax M2.7: Early Echoes of Self-Evolution",
      "authors": "MiniMax",
      "year": "2026",
      "tasks": "Multilingual software-engineering tasks",
      "format": "Repository task completion",
      "difficulty": "Professional software engineering",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 36,
      "url": "https://benchlm.ai/benchmarks/swe-bench-multilingual",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/swe-bench-multilingual.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "sweMultimodal",
      "name": "SWE Multimodal",
      "fullName": "SWE-bench Multimodal",
      "description": "A multimodal variant of SWE-bench that adds visual context such as screenshots and design mockups to software engineering issue descriptions.",
      "paperUrl": "https://www.swebench.com/multimodal",
      "paperTitle": "SWE-bench Multimodal",
      "authors": "SWE-bench team",
      "year": "2025",
      "tasks": "Multimodal software engineering tasks",
      "format": "Code patch generation with visual context",
      "difficulty": "Frontier multimodal coding",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 4,
      "url": "https://benchlm.ai/benchmarks/swe-bench-multimodal",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/swe-bench-multimodal.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "cursorBench",
      "name": "CursorBench",
      "fullName": "CursorBench",
      "description": "Cursor's current first-party benchmark for ambiguous, multi-file coding-agent tasks from real Cursor sessions.",
      "paperUrl": "https://cursor.com/evals",
      "paperTitle": "CursorBench 3.2",
      "authors": "Cursor",
      "year": "2026",
      "tasks": "Harder long-horizon agentic coding tasks",
      "format": "Cursor agent-loop evaluation",
      "difficulty": "Professional agentic software engineering",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/cursorbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/cursorbench.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "multiSweBench",
      "name": "Multi-SWE Bench",
      "fullName": "Multi-SWE Bench",
      "description": "A multi-language software-engineering benchmark that measures repository-level bug fixing and implementation across more than one programming ecosystem.",
      "paperUrl": "https://www.minimax.io/news/minimax-m27-en",
      "paperTitle": "MiniMax M2.7: Early Echoes of Self-Evolution",
      "authors": "MiniMax",
      "year": "2026",
      "tasks": "Multi-language repo tasks",
      "format": "Repository task completion",
      "difficulty": "Professional software engineering",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/multiswebench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/multiswebench.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "vibePro",
      "name": "VIBE-Pro",
      "fullName": "VIBE-Pro",
      "description": "A repo-level code generation and full-project delivery benchmark spanning web, mobile, and simulation-style implementation tasks.",
      "paperUrl": "https://www.minimax.io/news/minimax-m27-en",
      "paperTitle": "MiniMax M2.7: Early Echoes of Self-Evolution",
      "authors": "MiniMax",
      "year": "2026",
      "tasks": "Full project delivery tasks",
      "format": "Repository-level implementation benchmark",
      "difficulty": "End-to-end software delivery",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/vibepro",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/vibepro.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "vibeCodeBench",
      "name": "Vibe Code Bench",
      "fullName": "Vibe Code Bench v1.1",
      "description": "Vals.ai benchmark for evaluating whether models can build complete web applications from natural language specifications in a production-like development environment.",
      "paperUrl": "https://www.vals.ai/benchmarks/vibe-code",
      "paperTitle": "Vibe Code Bench: Evaluating AI Models on End-to-End Web Application Development",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "End-to-end web application builds",
      "format": "Full-stack app implementation benchmark",
      "difficulty": "End-to-end software delivery",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 41,
      "url": "https://benchlm.ai/benchmarks/vibecodebench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/vibecodebench.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "programBench",
      "name": "ProgramBench",
      "fullName": "ProgramBench: Can Language Models Rebuild Programs From Scratch?",
      "description": "A cleanroom software-engineering benchmark where agents receive only a compiled executable and documentation, then must architect and implement a complete codebase that reproduces the original program's behavior.",
      "paperUrl": "https://programbench.com/static/paper.pdf",
      "paperTitle": "ProgramBench: Can Language Models Rebuild Programs From Scratch?",
      "authors": "John Yang, Kilian Lieret, Jeffrey Ma, Parth Thakkar, Dmitrii Pedchenko, Sten Sootla, Emily McMilin, Pengcheng Yin, Rui Hou, Gabriel Synnaeve, Diyi Yang, Ofir Press",
      "year": "2026",
      "tasks": "200 program reconstruction tasks",
      "format": "Cleanroom executable reimplementation",
      "difficulty": "Full-repository software architecture",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 7,
      "url": "https://benchlm.ai/benchmarks/programbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/programbench.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "postTrainBench",
      "name": "PostTrain Bench",
      "fullName": "PostTrain Bench",
      "description": "A software-engineering benchmark for post-training infrastructure and implementation tasks, evaluated through the official Harbor implementation.",
      "paperUrl": "https://www.kimi.com/blog/kimi-k3",
      "paperTitle": "Kimi K3: Open Frontier Intelligence",
      "authors": "Moonshot AI",
      "year": "2026",
      "tasks": "Post-training software-engineering tasks",
      "format": "Harbor agent evaluation",
      "difficulty": "Frontier software engineering",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 3,
      "url": "https://benchlm.ai/benchmarks/posttrainbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/posttrainbench.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "frontierSwe",
      "name": "FrontierSWE",
      "fullName": "FrontierSWE",
      "description": "An ultra-long-horizon software-engineering benchmark with open-ended implementation, performance, and research tasks designed to challenge frontier coding agents.",
      "paperUrl": "https://www.frontierswe.com/blog",
      "paperTitle": "FrontierSWE: Benchmarking coding agents at the limits of human abilities",
      "authors": "Evan Chu, Rajan Agarwal, Abishek Thangamuthu, Brendan Graham, Justus Mattern",
      "year": "2026",
      "tasks": "17 ultra-long-horizon engineering and research tasks",
      "format": "Mean@5, best@5, average rank, and dominance",
      "difficulty": "Ultra-long-horizon frontier software engineering",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 3,
      "url": "https://benchlm.ai/benchmarks/frontierswe",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/frontierswe.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "kimiCodeBenchV2",
      "name": "Kimi Code Bench v2",
      "fullName": "Kimi Code Bench v2",
      "description": "A Moonshot AI internal coding-agent benchmark for realistic software-engineering tasks across mainstream programming languages and production technology stacks.",
      "paperUrl": "https://huggingface.co/moonshotai/Kimi-K2.7-Code",
      "paperTitle": "Kimi K2.7 Code",
      "authors": "Moonshot AI",
      "year": "2026",
      "tasks": "Realistic coding-agent tasks",
      "format": "Coding-agent pass rate",
      "difficulty": "Production software engineering",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/kimicodebenchv2",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/kimicodebenchv2.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "mlsBenchLite",
      "name": "MLS-Bench Lite",
      "fullName": "MLS-Bench Lite",
      "description": "A 30-task subset of MLS-Bench that evaluates whether AI systems can invent generalizable and scalable machine-learning methods.",
      "paperUrl": "https://mls-bench.com/",
      "paperTitle": "MLS-Bench",
      "authors": "MLS-Bench",
      "year": "2026",
      "tasks": "30 machine-learning research tasks",
      "format": "Agentic ML task evaluation",
      "difficulty": "ML research and systems engineering",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 3,
      "url": "https://benchlm.ai/benchmarks/mlsbenchlite",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mlsbenchlite.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "paperBench",
      "name": "PaperBench",
      "fullName": "PaperBench",
      "description": "A research-reproduction benchmark that asks agents to recreate the contributions of AI papers from the paper alone.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.8",
      "paperTitle": "Qwen3.8-Max: A New Bar for Coding and Cowork",
      "authors": "Qwen Team",
      "year": "2026",
      "tasks": "AI research-paper reproduction",
      "format": "Long-horizon agent evaluation",
      "difficulty": "Frontier autonomous research and engineering",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/paperbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/paperbench.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "qwenReactBench",
      "name": "QwenReactBench",
      "fullName": "QwenReactBench",
      "description": "Qwen's internal benchmark for building and rendering bilingual React projects across seven categories.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.8",
      "paperTitle": "Qwen3.8-Max: A New Bar for Coding and Cowork",
      "authors": "Qwen Team",
      "year": "2026",
      "tasks": "Bilingual React project construction",
      "format": "Bradley-Terry/Elo rating",
      "difficulty": "Production frontend development",
      "decimals": 0,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/qwenreactbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/qwenreactbench.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "nl2Repo",
      "name": "NL2Repo",
      "fullName": "NL2Repo",
      "description": "A repository-understanding benchmark that measures whether models can map natural-language requests onto the right code locations and system changes.",
      "paperUrl": "https://www.minimax.io/news/minimax-m27-en",
      "paperTitle": "MiniMax M2.7: Early Echoes of Self-Evolution",
      "authors": "MiniMax",
      "year": "2026",
      "tasks": "Natural language to repository tasks",
      "format": "Repository understanding benchmark",
      "difficulty": "System-level software comprehension",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 25,
      "url": "https://benchlm.ai/benchmarks/nl2repo",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/nl2repo.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "dsBenchFullStack",
      "name": "DSBench-FullStack",
      "fullName": "DeepSeek DSBench FullStack",
      "description": "DeepSeek's internal full-stack coding-agent benchmark.",
      "paperUrl": "https://api-docs.deepseek.com/zh-cn/updates/",
      "paperTitle": "DeepSeek V4 Flash 0731 update",
      "authors": "DeepSeek-AI",
      "year": "2026",
      "tasks": "Internal full-stack coding-agent tasks",
      "format": "Provider-reported score",
      "difficulty": "Full-stack software engineering",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/dsbenchfullstack",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/dsbenchfullstack.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "dsBenchHard",
      "name": "DSBench-Hard",
      "fullName": "DeepSeek DSBench Hard",
      "description": "DeepSeek's internal hard coding-agent benchmark.",
      "paperUrl": "https://api-docs.deepseek.com/zh-cn/updates/",
      "paperTitle": "DeepSeek V4 Flash 0731 update",
      "authors": "DeepSeek-AI",
      "year": "2026",
      "tasks": "Internal hard coding-agent tasks",
      "format": "Provider-reported score",
      "difficulty": "Advanced coding-agent challenges",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/dsbenchhard",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/dsbenchhard.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "reactNativeEvals",
      "name": "React Native Evals",
      "fullName": "React Native Evals",
      "description": "An open benchmark for AI coding agents on real-world React Native implementation tasks, emphasizing working app behavior, recommended architecture choices, and strict constraint adherence.",
      "paperUrl": "https://rn-evals.vercel.app/",
      "paperTitle": "React Native Evals",
      "authors": "Callstack",
      "year": "2026",
      "tasks": "React Native app implementation tasks",
      "format": "Framework-specific app development evaluation",
      "difficulty": "Production mobile app engineering",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 16,
      "url": "https://benchlm.ai/benchmarks/reactnativeevals",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/reactnativeevals.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "reactBench",
      "name": "ReactBench",
      "fullName": "ReactBench v1",
      "description": "A coding-agent benchmark for realistic React work, with rubrics that check production concerns such as performance, accessibility, correctness, and code quality.",
      "paperUrl": "https://www.reactbench.com/",
      "paperTitle": "ReactBench",
      "authors": "Million",
      "year": "2026",
      "tasks": "51 production React tasks",
      "format": "Pass@1 weighted rubric score",
      "difficulty": "Production frontend engineering",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/reactbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/reactbench.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "kernelBench",
      "name": "KernelBench",
      "fullName": "KernelBench Hard H100",
      "description": "An agentic GPU-kernel benchmark that measures how much of the hardware roofline a model's correct, audit-clean kernels reach on six demanding CUDA and Triton problems.",
      "paperUrl": "https://kernelbench.com/hard?gpu=h100",
      "paperTitle": "KernelBench Hard",
      "authors": "Elliot Arledge",
      "year": "2026",
      "tasks": "6 GPU-kernel optimization problems",
      "format": "Mean peak fraction of hardware roofline over valid cells",
      "difficulty": "Agentic GPU systems engineering",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/kernelbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/kernelbench.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "nextjsEvals",
      "name": "Next.js Evals",
      "fullName": "AI Agent Evaluations for Next.js",
      "description": "A Vercel benchmark for AI coding agents on Next.js code generation and migration tasks, reporting success rate, average execution time, and an AGENTS.md documentation-assisted split.",
      "paperUrl": "https://nextjs.org/evals",
      "paperTitle": "AI Agent Evaluations | Next.js",
      "authors": "Vercel",
      "year": "2026",
      "tasks": "24 Next.js code generation and migration tasks",
      "format": "Agent task completion with withheld Vitest assertions",
      "difficulty": "Framework-specific web application engineering",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/nextjsevals",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/nextjsevals.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "sweVerifiedArcee",
      "name": "SWE-bench Verified*",
      "fullName": "SWE-bench Verified (mini-swe-agent-v2)",
      "description": "A display-only SWE-bench Verified reference from Arcee AI's Trinity-Large-Thinking comparison chart.",
      "paperUrl": "https://www.arcee.ai/blog/trinity-large-thinking",
      "paperTitle": "Trinity-Large-Thinking: Scaling an Open Source Frontier Agent",
      "authors": "Arcee AI",
      "year": "2026",
      "tasks": "Repository task completion",
      "format": "Agent scaffold benchmark",
      "difficulty": "Professional software engineering",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 5,
      "url": "https://benchlm.ai/benchmarks/sweverifiedarcee",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/sweverifiedarcee.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "spider2Lite",
      "name": "Spider 2.0-Lite",
      "fullName": "Spider 2.0-Lite",
      "description": "A text-to-SQL benchmark over realistic warehouse-scale schemas, reported by Interfaze for model comparison.",
      "paperUrl": "https://github.com/xlang-ai/Spider2",
      "paperTitle": "Spider 2.0: Evaluating Language Models on Real-World Enterprise Text-to-SQL Workflows",
      "authors": "Spider 2.0 authors",
      "year": "2024",
      "tasks": "Text-to-SQL queries",
      "format": "Execution accuracy",
      "difficulty": "Enterprise text-to-SQL",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/spider2lite",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/spider2lite.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "sciCode",
      "name": "SciCode",
      "fullName": "Scientific Code Benchmark",
      "description": "SciCode evaluates language models on generating code for realistic scientific research problems across 16 subfields of physics, math, chemistry, biology, and material science. Problems decompose into 338 subproblems requiring domain knowledge recall, scientific reasoning, and precise code synthesis. Based on real scripts from published research.",
      "paperUrl": null,
      "paperTitle": null,
      "authors": null,
      "year": 2024,
      "tasks": 80,
      "format": null,
      "difficulty": null,
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": 0.16,
      "displayableScoreCount": 19,
      "url": "https://benchlm.ai/benchmarks/scicode",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/scicode.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "aaCodingIndex",
      "name": "AA Coding Index",
      "fullName": "Artificial Analysis Coding Index",
      "description": "A display-only Artificial Analysis coding index.",
      "paperUrl": "https://artificialanalysis.ai/leaderboards/models",
      "paperTitle": "Artificial Analysis model leaderboards",
      "authors": "Artificial Analysis",
      "year": "2026",
      "tasks": "Cross-benchmark coding index",
      "format": "Aggregated model score",
      "difficulty": "Display-only external reference",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 101,
      "url": "https://benchlm.ai/benchmarks/aacodingindex",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/aacodingindex.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "aaCodingAgents",
      "name": "AA Coding Agents",
      "fullName": "Artificial Analysis Coding Agent Index",
      "description": "A display-only Artificial Analysis leaderboard for coding-agent systems, combining agent harnesses, host models, and execution settings across software-engineering benchmarks.",
      "paperUrl": "https://artificialanalysis.ai/agents/coding-agents",
      "paperTitle": "Artificial Analysis Coding Agent Benchmarks",
      "authors": "Artificial Analysis",
      "year": "2026",
      "tasks": "Composite over DeepSWE, Terminal-Bench v2, and SWE-Atlas-QnA",
      "format": "Average pass@1 index",
      "difficulty": "Real-world coding-agent workflows",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/aacodingagents",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/aacodingagents.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "aaSciCode",
      "name": "AA-SciCode",
      "fullName": "Artificial Analysis SciCode",
      "description": "A display-only Artificial Analysis SciCode score.",
      "paperUrl": "https://artificialanalysis.ai/evaluations/scicode",
      "paperTitle": "Artificial Analysis SciCode Benchmark Leaderboard",
      "authors": "Artificial Analysis",
      "year": "2026",
      "tasks": "Scientific coding subproblems",
      "format": "Task success rate",
      "difficulty": "Scientific programming",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 165,
      "url": "https://benchlm.ai/benchmarks/aascicode",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/aascicode.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "terminalBenchHard",
      "name": "Terminal-Bench Hard",
      "fullName": "Terminal-Bench Hard",
      "description": "A display-only Artificial Analysis coding metric for agentic coding and terminal use on a harder Terminal-Bench slice.",
      "paperUrl": "https://artificialanalysis.ai/models/grok-4-3",
      "paperTitle": "Artificial Analysis model benchmarks",
      "authors": "Artificial Analysis",
      "year": "2026",
      "tasks": "Agentic coding and terminal tasks",
      "format": "Task success rate",
      "difficulty": "Professional software engineering",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/terminal-bench-hard",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/terminal-bench-hard.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "vibeV2",
      "name": "VIBE V2",
      "fullName": "VIBE V2",
      "description": "A display-only MiniMax provider benchmark for end-to-end coding-agent and product-building tasks.",
      "paperUrl": "https://huggingface.co/MiniMaxAI/MiniMax-M3",
      "paperTitle": "MiniMax M3 model card",
      "authors": "MiniMax",
      "year": "2026",
      "tasks": "End-to-end coding-agent tasks",
      "format": "Task success rate",
      "difficulty": "Frontier coding-agent workflows",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/vibev2",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/vibev2.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "svgBench",
      "name": "SVG-Bench",
      "fullName": "SVG-Bench",
      "description": "A display-only provider benchmark for generating or manipulating SVG outputs from natural-language requirements.",
      "paperUrl": "https://huggingface.co/MiniMaxAI/MiniMax-M3",
      "paperTitle": "MiniMax M3 model card",
      "authors": "MiniMax",
      "year": "2026",
      "tasks": "SVG generation and editing tasks",
      "format": "Task success rate",
      "difficulty": "Visual coding and structured graphics generation",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/svgbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/svgbench.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "kernelBenchHard",
      "name": "KernelBench Hard",
      "fullName": "KernelBench Hard",
      "description": "A display-only benchmark for difficult GPU kernel implementation and optimization tasks.",
      "paperUrl": "https://huggingface.co/MiniMaxAI/MiniMax-M3",
      "paperTitle": "MiniMax M3 model card",
      "authors": "MiniMax",
      "year": "2026",
      "tasks": "Hard GPU kernel coding tasks",
      "format": "Task success rate",
      "difficulty": "Specialized systems programming",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/kernelbenchhard",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/kernelbenchhard.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "gameDevBench",
      "name": "GameDevBench",
      "fullName": "GameDevBench",
      "description": "Evaluates coding agents on 333 multimodal game-development tasks in Godot, spanning 2D graphics, 3D graphics, user interfaces, and gameplay logic.",
      "paperUrl": "https://arxiv.org/abs/2602.11103",
      "paperTitle": "GameDevBench: Evaluating Agentic Capabilities Through Game Development",
      "authors": "Wayne Chi, Yixiong Fang, Arnav Yayavaram, Siddharth Yayavaram, Seth Karten, Qiuhong Anna Wei, Runkun Chen, Alexander Wang, Valerie Chen, Ameet Talwalkar, and Chris Donahue",
      "year": "2026",
      "tasks": "333 tasks from 88 tutorials",
      "format": "Pass@1 on the full task set with 95% confidence intervals",
      "difficulty": "Multimodal game development in Godot 4.4.1",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/gamedevbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/gamedevbench.md"
    },
    {
      "category": "coding",
      "categoryLabel": "Coding",
      "benchmarkKey": "edgeBench",
      "name": "EdgeBench",
      "fullName": "EdgeBench",
      "description": "A systems and software-engineering benchmark from ByteDance Seed that evaluates agents on long-horizon edge tasks using time-budgeted learning curves rather than a single static pass rate.",
      "paperUrl": "https://edge-bench.org/",
      "paperTitle": "EdgeBench",
      "authors": "ByteDance Seed",
      "year": "2026",
      "tasks": "Systems and software-engineering tasks",
      "format": "Time-budgeted agent learning curves",
      "difficulty": "Long-horizon engineering",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/edgebench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/edgebench.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "imo2026",
      "name": "IMO 2026",
      "fullName": "International Mathematical Olympiad 2026",
      "description": "Proof-based olympiad performance on all six IMO 2026 problems.",
      "paperUrl": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",
      "paperTitle": "Claude Opus 5 System Card",
      "authors": "Anthropic",
      "year": "2026",
      "tasks": "6 proof-based problems",
      "format": "Official-style proof score",
      "difficulty": "International olympiad mathematics",
      "decimals": 0,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/imo2026",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/imo2026.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "riemannBench",
      "name": "RiemannBench (no tools)",
      "fullName": "RiemannBench without tools",
      "description": "Research-level mathematics problems with unique programmatically verified closed-form answers.",
      "paperUrl": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",
      "paperTitle": "Claude Opus 5 System Card",
      "authors": "Surge AI and Anthropic",
      "year": "2026",
      "tasks": "25 private research-level mathematics problems",
      "format": "Accuracy without tools",
      "difficulty": "Research mathematics",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/riemannbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/riemannbench.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "riemannBenchWithTools",
      "name": "RiemannBench (tools)",
      "fullName": "RiemannBench with tools",
      "description": "Research-level mathematics problems with unique programmatically verified closed-form answers and tool access.",
      "paperUrl": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",
      "paperTitle": "Claude Opus 5 System Card",
      "authors": "Surge AI and Anthropic",
      "year": "2026",
      "tasks": "25 private research-level mathematics problems",
      "format": "Accuracy with tools",
      "difficulty": "Research mathematics",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/riemannbenchwithtools",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/riemannbenchwithtools.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "arxivMathJune2026",
      "name": "ArXivMath Jun. 2026 (no tools)",
      "fullName": "ArXivMath June 2026 without tools",
      "description": "Final-answer research mathematics problems drawn from recent arXiv abstracts.",
      "paperUrl": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",
      "paperTitle": "Claude Opus 5 System Card",
      "authors": "MathArena and Anthropic",
      "year": "2026",
      "tasks": "49 recent research-mathematics problems",
      "format": "Final-answer accuracy",
      "difficulty": "Research mathematics",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/arxivmathjune2026",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/arxivmathjune2026.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "arxivMathJune2026WithTools",
      "name": "ArXivMath Jun. 2026 (tools)",
      "fullName": "ArXivMath June 2026 with tools",
      "description": "Final-answer research mathematics problems drawn from recent arXiv abstracts with tool access.",
      "paperUrl": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",
      "paperTitle": "Claude Opus 5 System Card",
      "authors": "MathArena and Anthropic",
      "year": "2026",
      "tasks": "49 recent research-mathematics problems",
      "format": "Final-answer accuracy with tools",
      "difficulty": "Research mathematics",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/arxivmathjune2026withtools",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/arxivmathjune2026withtools.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "aaAime2025",
      "name": "AA AIME 2025",
      "fullName": "Artificial Analysis AIME 2025",
      "description": "An independently evaluated AIME 2025 result from Artificial Analysis.",
      "paperUrl": "https://artificialanalysis.ai/evaluations/aime-2025",
      "paperTitle": "Artificial Analysis AIME 2025 Benchmark Leaderboard",
      "authors": "Artificial Analysis",
      "year": "2026",
      "tasks": "30 AIME 2025 problems",
      "format": "Accuracy",
      "difficulty": "Olympiad mathematics",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/aaaime2025",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/aaaime2025.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "aaMath500",
      "name": "AA MATH-500",
      "fullName": "Artificial Analysis MATH-500",
      "description": "An independently evaluated MATH-500 result from Artificial Analysis.",
      "paperUrl": "https://artificialanalysis.ai/evaluations/math-500",
      "paperTitle": "Artificial Analysis MATH-500 Benchmark Leaderboard",
      "authors": "Artificial Analysis",
      "year": "2026",
      "tasks": "500 competition mathematics problems",
      "format": "Accuracy",
      "difficulty": "High school to undergraduate mathematics",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 3,
      "url": "https://benchlm.ai/benchmarks/aamath500",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/aamath500.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "aime2023",
      "name": "AIME 2023",
      "fullName": "American Invitational Mathematics Examination 2023",
      "description": "A 15-question, 3-hour examination where each answer is an integer from 000 to 999. Serves as the intermediate step between AMC 10/12 and the USA Mathematical Olympiad (USAMO).",
      "paperUrl": "https://www.maa.org/math-competitions/aime",
      "paperTitle": "American Invitational Mathematics Examination",
      "authors": "Mathematical Association of America",
      "year": "2023",
      "tasks": "15 problems",
      "format": "Integer answers 000-999",
      "difficulty": "High school olympiad level",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/aime2023",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/aime2023.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "aime2024",
      "name": "AIME 2024",
      "fullName": "American Invitational Mathematics Examination 2024",
      "description": "The 2024 edition of AIME, maintaining the same format of 15 challenging mathematics problems with integer answers from 000 to 999.",
      "paperUrl": "https://www.maa.org/math-competitions/aime",
      "paperTitle": "American Invitational Mathematics Examination",
      "authors": "Mathematical Association of America",
      "year": "2024",
      "tasks": "15 problems",
      "format": "Integer answers 000-999",
      "difficulty": "High school olympiad level",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/aime2024",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/aime2024.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "aime2025",
      "name": "AIME 2025",
      "fullName": "American Invitational Mathematics Examination 2025",
      "description": "The most recent AIME examination, featuring 15 challenging mathematics problems testing olympiad-level mathematical reasoning with integer answers from 000-999.",
      "paperUrl": "https://www.maa.org/math-competitions/aime",
      "paperTitle": "American Invitational Mathematics Examination",
      "authors": "Mathematical Association of America",
      "year": "2025",
      "tasks": "15 problems",
      "format": "Integer answers 000-999",
      "difficulty": "High school olympiad level",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 12,
      "url": "https://benchlm.ai/benchmarks/aime2025",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/aime2025.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "gsm8k",
      "name": "GSM8K",
      "fullName": "Grade School Math 8K",
      "description": "A grade-school mathematical reasoning benchmark reported in DeepSeek-V4 base-model evaluations.",
      "paperUrl": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf",
      "paperTitle": "DeepSeek-V4 Technical Report",
      "authors": "DeepSeek-AI",
      "year": "2026",
      "tasks": "Grade-school math word problems",
      "format": "Exact match",
      "difficulty": "Grade-school math",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/gsm8k",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/gsm8k.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "mathBenchmark",
      "name": "MATH",
      "fullName": "MATH",
      "description": "A competition-style mathematical reasoning benchmark reported in DeepSeek-V4 base-model evaluations.",
      "paperUrl": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf",
      "paperTitle": "DeepSeek-V4 Technical Report",
      "authors": "DeepSeek-AI",
      "year": "2026",
      "tasks": "Competition math problems",
      "format": "Exact match",
      "difficulty": "Advanced math reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/mathbenchmark",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mathbenchmark.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "cmath",
      "name": "CMath",
      "fullName": "CMath",
      "description": "A Chinese mathematical reasoning benchmark reported in DeepSeek-V4 base-model evaluations.",
      "paperUrl": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf",
      "paperTitle": "DeepSeek-V4 Technical Report",
      "authors": "DeepSeek-AI",
      "year": "2026",
      "tasks": "Chinese math problems",
      "format": "Exact match",
      "difficulty": "Math reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/cmath",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/cmath.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "aime2025Arcee",
      "name": "AIME25 (Arcee)",
      "fullName": "AIME25 first-party comparison snapshot",
      "description": "A display-only AIME25 reference from Arcee AI's Trinity-Large-Thinking launch chart.",
      "paperUrl": "https://www.arcee.ai/blog/trinity-large-thinking",
      "paperTitle": "Trinity-Large-Thinking: Scaling an Open Source Frontier Agent",
      "authors": "Arcee AI",
      "year": "2026",
      "tasks": "15 problems",
      "format": "Integer answers 000-999",
      "difficulty": "High school olympiad level",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 6,
      "url": "https://benchlm.ai/benchmarks/aime2025arcee",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/aime2025arcee.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "hmmt2023",
      "name": "HMMT Feb 2023",
      "fullName": "Harvard-MIT Mathematics Tournament February 2023",
      "description": "A prestigious high school mathematics competition hosted jointly by Harvard and MIT, featuring challenging problems across various mathematical disciplines.",
      "paperUrl": "https://www.hmmt.org/",
      "paperTitle": "Harvard-MIT Mathematics Tournament",
      "authors": "Harvard and MIT Mathematics Departments",
      "year": "2023",
      "tasks": "Tournament problems",
      "format": "Competition mathematics",
      "difficulty": "High school olympiad level",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/hmmt2023",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/hmmt2023.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "hmmt2024",
      "name": "HMMT Feb 2024",
      "fullName": "Harvard-MIT Mathematics Tournament February 2024",
      "description": "The 2024 February edition of the Harvard-MIT Mathematics Tournament, continuing the tradition of challenging high school mathematics competition.",
      "paperUrl": "https://www.hmmt.org/",
      "paperTitle": "Harvard-MIT Mathematics Tournament",
      "authors": "Harvard and MIT Mathematics Departments",
      "year": "2024",
      "tasks": "Tournament problems",
      "format": "Competition mathematics",
      "difficulty": "High school olympiad level",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/hmmt2024",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/hmmt2024.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "hmmt2025",
      "name": "HMMT Feb 2025",
      "fullName": "Harvard-MIT Mathematics Tournament February 2025",
      "description": "The most recent February edition of the Harvard-MIT Mathematics Tournament, featuring the latest challenging problems in competitive mathematics.",
      "paperUrl": "https://www.hmmt.org/",
      "paperTitle": "Harvard-MIT Mathematics Tournament",
      "authors": "Harvard and MIT Mathematics Departments",
      "year": "2025",
      "tasks": "Tournament problems",
      "format": "Competition mathematics",
      "difficulty": "High school olympiad level",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/hmmt2025",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/hmmt2025.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "brumo2025",
      "name": "BRUMO 2025",
      "fullName": "Bulgarian Mathematical Olympiad 2025",
      "description": "A challenging mathematical olympiad competition featuring problems that test advanced mathematical reasoning and problem-solving skills at the olympiad level.",
      "paperUrl": "https://www.math.bas.bg/",
      "paperTitle": "Bulgarian Mathematical Olympiad",
      "authors": "Bulgarian Mathematical Society",
      "year": "2025",
      "tasks": "Olympiad problems",
      "format": "Mathematical olympiad",
      "difficulty": "Mathematical olympiad level",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/brumo2025",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/brumo2025.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "math500",
      "name": "MATH-500",
      "fullName": "MATH-500 Problem Set",
      "description": "A curated subset of 500 problems from the MATH dataset, covering algebra, counting and probability, geometry, intermediate algebra, number theory, prealgebra, and precalculus.",
      "paperUrl": "https://arxiv.org/abs/2103.03874",
      "paperTitle": "Measuring Mathematical Problem Solving With the MATH Dataset",
      "authors": "Dan Hendrycks, Collin Burns, Saurav Kadavath, Akul Arora, Steven Basart, Eric Tang, Dawn Song, Jacob Steinhardt",
      "year": "2021",
      "tasks": "500 problems",
      "format": "Free-form mathematical answers",
      "difficulty": "High school to undergraduate",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/math-500",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/math-500.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "aime2026",
      "name": "AIME26",
      "fullName": "AIME 2026",
      "description": "A 2026 American Invitational Mathematics Examination snapshot used in frontier-model comparison tables for mathematical reasoning.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2026",
      "tasks": "Competition math problems",
      "format": "Short-answer mathematics",
      "difficulty": "Olympiad-style mathematics",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": 0.25,
      "displayableScoreCount": 20,
      "url": "https://benchlm.ai/benchmarks/aime2026",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/aime2026.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "ipho2025Theory",
      "name": "IPhO 2025 (Theory)",
      "fullName": "International Physics Olympiad 2025 (Theory)",
      "description": "The three official theory problems from the 2025 International Physics Olympiad, scored with blinded human evaluation.",
      "paperUrl": "https://ai.meta.com/static-resource/muse-spark-eval-methodology",
      "paperTitle": "Muse Spark Eval Methodology",
      "authors": "Meta AI",
      "year": "2026",
      "tasks": "3 olympiad theory problems",
      "format": "Physics olympiad theory",
      "difficulty": "International olympiad physics",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/ipho2025theory",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/ipho2025theory.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "hmmtFeb2025",
      "name": "HMMT Feb 2025",
      "fullName": "Harvard-MIT Mathematics Tournament February 2025",
      "description": "A February 2025 HMMT slice used in exact-value provider tables for advanced contest-math reasoning.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2025",
      "tasks": "Competition math problems",
      "format": "Contest mathematics",
      "difficulty": "Olympiad-style mathematics",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 8,
      "url": "https://benchlm.ai/benchmarks/hmmtfeb2025",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/hmmtfeb2025.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "hmmtNov2025",
      "name": "HMMT Nov 2025",
      "fullName": "Harvard-MIT Mathematics Tournament November 2025",
      "description": "A November 2025 HMMT slice for high-end mathematical reasoning comparisons.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2025",
      "tasks": "Competition math problems",
      "format": "Contest mathematics",
      "difficulty": "Olympiad-style mathematics",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 9,
      "url": "https://benchlm.ai/benchmarks/hmmtnov2025",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/hmmtnov2025.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "hmmtFeb2026",
      "name": "HMMT Feb 2026",
      "fullName": "Harvard-MIT Mathematics Tournament February 2026",
      "description": "A February 2026 HMMT slice used in newer frontier-model math comparisons.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2026",
      "tasks": "Competition math problems",
      "format": "Contest mathematics",
      "difficulty": "Olympiad-style mathematics",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": 0.25,
      "displayableScoreCount": 19,
      "url": "https://benchlm.ai/benchmarks/hmmtfeb2026",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/hmmtfeb2026.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "imoAnswerBench",
      "name": "IMOAnswerBench",
      "fullName": "IMOAnswerBench",
      "description": "A challenging mathematical reasoning benchmark reported in DeepSeek-V4 model evaluations.",
      "paperUrl": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf",
      "paperTitle": "DeepSeek-V4 Technical Report",
      "authors": "DeepSeek-AI",
      "year": "2026",
      "tasks": "Advanced mathematical answer generation",
      "format": "Pass@1 math benchmark",
      "difficulty": "Olympiad-level mathematics",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 7,
      "url": "https://benchlm.ai/benchmarks/imoanswerbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/imoanswerbench.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "apex",
      "name": "Apex",
      "fullName": "Apex",
      "description": "A high-difficulty mathematical reasoning benchmark reported in DeepSeek-V4 model evaluations.",
      "paperUrl": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf",
      "paperTitle": "DeepSeek-V4 Technical Report",
      "authors": "DeepSeek-AI",
      "year": "2026",
      "tasks": "Advanced mathematical reasoning",
      "format": "Pass@1 math benchmark",
      "difficulty": "Frontier math reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 6,
      "url": "https://benchlm.ai/benchmarks/apex",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/apex.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "apexShortlist",
      "name": "Apex Shortlist",
      "fullName": "Apex Shortlist",
      "description": "A shortlist subset of the Apex mathematical reasoning benchmark reported in DeepSeek-V4 model evaluations.",
      "paperUrl": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf",
      "paperTitle": "DeepSeek-V4 Technical Report",
      "authors": "DeepSeek-AI",
      "year": "2026",
      "tasks": "Advanced mathematical reasoning",
      "format": "Pass@1 math benchmark",
      "difficulty": "Frontier math reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/apexshortlist",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/apexshortlist.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "mmAnswerBench",
      "name": "MMAnswerBench",
      "fullName": "MMAnswerBench",
      "description": "A multimodal mathematical reasoning benchmark that tests whether models can answer visually grounded math questions correctly.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2026",
      "tasks": "Multimodal math questions",
      "format": "Visual and structured mathematical QA",
      "difficulty": "Advanced mathematical reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 10,
      "url": "https://benchlm.ai/benchmarks/mmanswerbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mmanswerbench.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "frontierMath",
      "name": "FrontierMath (legacy)",
      "fullName": "FrontierMath legacy aggregate",
      "description": "Legacy FrontierMath values retained for historical model pages. This field is not used in current rankings because it can mix prior benchmark versions and slices.",
      "paperUrl": "https://epoch.ai/frontiermath",
      "paperTitle": "FrontierMath: A Benchmark for Evaluating Advanced Mathematical Reasoning in AI",
      "authors": "Epoch AI",
      "year": "2024",
      "tasks": "Historical aggregate",
      "format": "Open-ended mathematical reasoning with tool access",
      "difficulty": "Research-level mathematics",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 7,
      "url": "https://benchlm.ai/benchmarks/frontiermath",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/frontiermath.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "frontierMathV2Tiers13",
      "name": "FrontierMath v2 (Tiers 1-3)",
      "fullName": "FrontierMath v2 Tiers 1-3",
      "description": "Epoch AI's corrected v2 core FrontierMath suite of private advanced mathematics problems. Models can reason iteratively and use Python; scores are pass rates on the private set.",
      "paperUrl": "https://epoch.ai/benchmarks/frontiermath-tier-4-v2",
      "paperTitle": "FrontierMath v2 benchmark hub",
      "authors": "Epoch AI",
      "year": "2026",
      "tasks": "295 private advanced mathematics problems",
      "format": "Python-enabled iterative mathematical problem solving",
      "difficulty": "From olympiad-plus to early research mathematics",
      "decimals": 3,
      "successorKey": null,
      "successorUrl": null,
      "weight": 0.3,
      "displayableScoreCount": 51,
      "url": "https://benchlm.ai/benchmarks/frontiermathv2tiers13",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/frontiermathv2tiers13.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "frontierMathV2Tier4",
      "name": "FrontierMath v2 (Tier 4)",
      "fullName": "FrontierMath v2 Tier 4",
      "description": "Epoch AI's corrected v2 Tier 4 expansion, a separate set of exceptionally difficult research-level mathematics problems evaluated with Python-enabled iterative reasoning.",
      "paperUrl": "https://epoch.ai/benchmarks/frontiermath-tier-4-v2?view=graph&tab=leaderboard",
      "paperTitle": "FrontierMath Tier 4 v2 leaderboard",
      "authors": "Epoch AI",
      "year": "2026",
      "tasks": "43 private extreme-difficulty mathematics problems",
      "format": "Python-enabled iterative mathematical problem solving",
      "difficulty": "Research-level mathematics requiring hours or days of expert work",
      "decimals": 3,
      "successorKey": null,
      "successorUrl": null,
      "weight": 0.1,
      "displayableScoreCount": 44,
      "url": "https://benchlm.ai/benchmarks/frontiermathv2tier4",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/frontiermathv2tier4.md"
    },
    {
      "category": "math",
      "categoryLabel": "Mathematics",
      "benchmarkKey": "usamo2026",
      "name": "USAMO 2026",
      "fullName": "United States of America Mathematical Olympiad 2026",
      "description": "The premier US mathematical olympiad competition, featuring proof-based problems that require deep mathematical insight and rigorous argumentation at the highest competition level.",
      "paperUrl": "https://www.maa.org/math-competitions/usamo",
      "paperTitle": "United States of America Mathematical Olympiad",
      "authors": "Mathematical Association of America",
      "year": "2026",
      "tasks": "6 proof-based problems",
      "format": "Mathematical proof construction",
      "difficulty": "International olympiad level",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": 0.1,
      "displayableScoreCount": 3,
      "url": "https://benchlm.ai/benchmarks/usamo2026",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/usamo2026.md"
    },
    {
      "category": "reasoning",
      "categoryLabel": "Reasoning",
      "benchmarkKey": "arcAgi1",
      "name": "ARC-AGI-1",
      "fullName": "ARC-AGI-1 Semi-Private Evaluation",
      "description": "ARC Prize fluid-intelligence benchmark using novel visual grid transformations.",
      "paperUrl": "https://arcprize.org/",
      "paperTitle": "ARC Prize leaderboard",
      "authors": "ARC Prize Foundation",
      "year": "2026",
      "tasks": "Semi-private ARC-AGI-1 evaluation set",
      "format": "Verified accuracy",
      "difficulty": "Abstract visual reasoning",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/arcagi1",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/arcagi1.md"
    },
    {
      "category": "reasoning",
      "categoryLabel": "Reasoning",
      "benchmarkKey": "musr",
      "name": "MuSR",
      "fullName": "Testing the Limits of Chain-of-thought with Multistep Soft Reasoning",
      "description": "A dataset for evaluating language models on multistep soft reasoning tasks specified in natural language narratives. Tests the ability to perform complex, structured reasoning.",
      "paperUrl": "https://arxiv.org/abs/2310.16049",
      "paperTitle": "MuSR: Testing the Limits of Chain-of-thought with Multistep Soft Reasoning",
      "authors": "Zayne Sprague, Xi Ye, Kaj Bostrom, Swarat Chaudhuri, Greg Durrett",
      "year": "2023",
      "tasks": "Multi-step reasoning",
      "format": "Narrative-based reasoning",
      "difficulty": "Complex reasoning tasks",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/musr",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/musr.md"
    },
    {
      "category": "reasoning",
      "categoryLabel": "Reasoning",
      "benchmarkKey": "bbh",
      "name": "BBH",
      "fullName": "BIG-Bench Hard",
      "description": "A suite of 23 challenging tasks from the BIG-Bench collaborative benchmark where prior language models failed to exceed average human performance, even with chain-of-thought prompting.",
      "paperUrl": "https://arxiv.org/abs/2210.09261",
      "paperTitle": "Challenging BIG-Bench Tasks and Whether Chain-of-Thought Can Solve Them",
      "authors": "Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei",
      "year": "2022",
      "tasks": "23 tasks",
      "format": "Mixed reasoning tasks",
      "difficulty": "Advanced reasoning",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 3,
      "url": "https://benchlm.ai/benchmarks/bbh",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/bbh.md"
    },
    {
      "category": "reasoning",
      "categoryLabel": "Reasoning",
      "benchmarkKey": "drop",
      "name": "DROP",
      "fullName": "Discrete Reasoning Over Paragraphs",
      "description": "A reading-comprehension benchmark requiring discrete reasoning over paragraphs, reported in DeepSeek-V4 base-model evaluations.",
      "paperUrl": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf",
      "paperTitle": "DeepSeek-V4 Technical Report",
      "authors": "DeepSeek-AI",
      "year": "2026",
      "tasks": "Paragraph reasoning questions",
      "format": "F1",
      "difficulty": "Reading and numerical reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/drop",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/drop.md"
    },
    {
      "category": "reasoning",
      "categoryLabel": "Reasoning",
      "benchmarkKey": "hellaswag",
      "name": "HellaSwag",
      "fullName": "HellaSwag",
      "description": "A commonsense natural-language inference benchmark reported in DeepSeek-V4 base-model evaluations.",
      "paperUrl": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf",
      "paperTitle": "DeepSeek-V4 Technical Report",
      "authors": "DeepSeek-AI",
      "year": "2026",
      "tasks": "Commonsense completion questions",
      "format": "Exact match",
      "difficulty": "Commonsense reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/hellaswag",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/hellaswag.md"
    },
    {
      "category": "reasoning",
      "categoryLabel": "Reasoning",
      "benchmarkKey": "winogrande",
      "name": "WinoGrande",
      "fullName": "WinoGrande",
      "description": "A commonsense coreference benchmark reported in DeepSeek-V4 base-model evaluations.",
      "paperUrl": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf",
      "paperTitle": "DeepSeek-V4 Technical Report",
      "authors": "DeepSeek-AI",
      "year": "2026",
      "tasks": "Coreference resolution questions",
      "format": "Exact match",
      "difficulty": "Commonsense reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/winogrande",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/winogrande.md"
    },
    {
      "category": "reasoning",
      "categoryLabel": "Reasoning",
      "benchmarkKey": "cluewsc",
      "name": "CLUEWSC",
      "fullName": "CLUEWSC",
      "description": "A Chinese Winograd Schema Challenge benchmark reported in DeepSeek-V4 base-model evaluations.",
      "paperUrl": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf",
      "paperTitle": "DeepSeek-V4 Technical Report",
      "authors": "DeepSeek-AI",
      "year": "2026",
      "tasks": "Chinese coreference questions",
      "format": "Exact match",
      "difficulty": "Chinese commonsense reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/cluewsc",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/cluewsc.md"
    },
    {
      "category": "reasoning",
      "categoryLabel": "Reasoning",
      "benchmarkKey": "lisanBench",
      "name": "LisanBench",
      "fullName": "LisanBench",
      "description": "A word-chain reasoning benchmark that tests planning, recall, constraint following, and vocabulary depth by asking models to extend non-repeating edit-distance-1 chains.",
      "paperUrl": "https://lisanbench.com/?tab=about",
      "paperTitle": "LisanBench methodology",
      "authors": "voice-from-the-outer-world",
      "year": "2026",
      "tasks": "50 starting words × 3 trials",
      "format": "Difficulty-weighted word-chain reasoning",
      "difficulty": "Open-ended lexical planning",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/lisanbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/lisanbench.md"
    },
    {
      "category": "reasoning",
      "categoryLabel": "Reasoning",
      "benchmarkKey": "conceptualReasoning",
      "name": "Conceptual Reasoning",
      "fullName": "Conceptual Reasoning Benchmark",
      "description": "Tests whether model judgments rank argumentative critiques in the same order as expert human ratings across philosophy, AI alignment, and other concept-heavy texts.",
      "paperUrl": "https://www.andrew.cmu.edu/user/coesterh/conceptual_reasoning_benchmark.html",
      "paperTitle": "Conceptual Reasoning Benchmark Results",
      "authors": "Emery Cooper, Caspar Oesterheld, Chi Nguyen, and Ethan Perez",
      "year": null,
      "tasks": "224 texts and 608 within-text critique pairs",
      "format": "Average pairwise-ranking loss against expert ratings",
      "difficulty": "Fuzzy, expert-rated argumentative reasoning",
      "decimals": 3,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/conceptual-reasoning",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/conceptual-reasoning.md"
    },
    {
      "category": "reasoning",
      "categoryLabel": "Reasoning",
      "benchmarkKey": "ppBench",
      "name": "Pencil Puzzle Bench",
      "fullName": "Pencil Puzzle Bench",
      "description": "A multi-step verifiable reasoning benchmark that evaluates whether models can solve pencil puzzles with unique solutions.",
      "paperUrl": "https://arxiv.org/abs/2603.02119",
      "paperTitle": "Pencil Puzzle Bench",
      "authors": "Approximate Labs",
      "year": "2026",
      "tasks": "300 evaluation puzzles",
      "format": "Direct and agentic puzzle solve rate",
      "difficulty": "Multi-step verifiable reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/ppbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/ppbench.md"
    },
    {
      "category": "reasoning",
      "categoryLabel": "Reasoning",
      "benchmarkKey": "longBenchV2",
      "name": "LongBench v2",
      "fullName": "LongBench v2",
      "description": "A long-context benchmark that measures whether models can actually use extended context windows for reasoning and retrieval.",
      "paperUrl": "https://arxiv.org/abs/2412.15204",
      "paperTitle": "LongBench v2",
      "authors": "LongBench v2 authors",
      "year": "2025",
      "tasks": "Long-context tasks",
      "format": "Extended-context retrieval and reasoning",
      "difficulty": "Hard long-context",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": 0.38,
      "displayableScoreCount": 11,
      "url": "https://benchlm.ai/benchmarks/longbench-v2",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/longbench-v2.md"
    },
    {
      "category": "reasoning",
      "categoryLabel": "Reasoning",
      "benchmarkKey": "mrcrv2",
      "name": "MRCRv2",
      "fullName": "MRCRv2",
      "description": "A long-context benchmark for memory, retrieval, and multi-round coherence over large contexts.",
      "paperUrl": "https://openai.com/index/introducing-gpt-5-2/",
      "paperTitle": "Introducing GPT-5.2 and GPT-5.2 Pro",
      "authors": "OpenAI",
      "year": "2025",
      "tasks": "Long-context retrieval",
      "format": "Multi-round long-context evaluation",
      "difficulty": "Hard long-context",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": 0.31,
      "displayableScoreCount": 9,
      "url": "https://benchlm.ai/benchmarks/mrcrv2",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mrcrv2.md"
    },
    {
      "category": "reasoning",
      "categoryLabel": "Reasoning",
      "benchmarkKey": "mrcrv2_64_128",
      "name": "MRCR v2 64K-128K",
      "fullName": "OpenAI MRCR v2 8-needle 64K-128K",
      "description": "MRCR v2 slice focused on long-context retrieval at 64K-128K lengths.",
      "paperUrl": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
      "paperTitle": "Introducing GPT-5.4 mini and nano",
      "authors": "OpenAI",
      "year": "2026",
      "tasks": "8-needle retrieval tasks",
      "format": "Long-context retrieval",
      "difficulty": "Long-context reasoning",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/mrcr-v2-64k-128k",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mrcr-v2-64k-128k.md"
    },
    {
      "category": "reasoning",
      "categoryLabel": "Reasoning",
      "benchmarkKey": "mrcrv2_128_256",
      "name": "MRCR v2 128K-256K",
      "fullName": "OpenAI MRCR v2 8-needle 128K-256K",
      "description": "MRCR v2 slice focused on very long contexts at 128K-256K lengths.",
      "paperUrl": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
      "paperTitle": "Introducing GPT-5.4 mini and nano",
      "authors": "OpenAI",
      "year": "2026",
      "tasks": "8-needle retrieval tasks",
      "format": "Very-long-context retrieval",
      "difficulty": "Very long-context reasoning",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/mrcr-v2-128k-256k",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mrcr-v2-128k-256k.md"
    },
    {
      "category": "reasoning",
      "categoryLabel": "Reasoning",
      "benchmarkKey": "graphwalksBfs128k",
      "name": "Graphwalks BFS 128K",
      "fullName": "Graphwalks BFS 0K-128K",
      "description": "Long-context graph traversal benchmark using breadth-first search tasks.",
      "paperUrl": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
      "paperTitle": "Introducing GPT-5.4 mini and nano",
      "authors": "OpenAI",
      "year": "2026",
      "tasks": "Graph traversal tasks",
      "format": "Long-context graph reasoning",
      "difficulty": "Algorithmic long-context reasoning",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/graphwalksbfs128k",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/graphwalksbfs128k.md"
    },
    {
      "category": "reasoning",
      "categoryLabel": "Reasoning",
      "benchmarkKey": "graphwalksParents128k",
      "name": "Graphwalks Parents 128K",
      "fullName": "Graphwalks parents 0-128K",
      "description": "Long-context benchmark for recovering parent relationships inside graph tasks.",
      "paperUrl": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
      "paperTitle": "Introducing GPT-5.4 mini and nano",
      "authors": "OpenAI",
      "year": "2026",
      "tasks": "Graph parent-retrieval tasks",
      "format": "Long-context graph reasoning",
      "difficulty": "Algorithmic long-context reasoning",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/graphwalksparents128k",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/graphwalksparents128k.md"
    },
    {
      "category": "reasoning",
      "categoryLabel": "Reasoning",
      "benchmarkKey": "mrcr1m",
      "name": "MRCR 1M",
      "fullName": "MRCR 1M",
      "description": "A million-token MRCR long-context retrieval benchmark reported in DeepSeek-V4 model evaluations.",
      "paperUrl": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf",
      "paperTitle": "DeepSeek-V4 Technical Report",
      "authors": "DeepSeek-AI",
      "year": "2026",
      "tasks": "Million-token retrieval",
      "format": "Long-context retrieval MMR",
      "difficulty": "Million-token long context",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 4,
      "url": "https://benchlm.ai/benchmarks/mrcr1m",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mrcr1m.md"
    },
    {
      "category": "reasoning",
      "categoryLabel": "Reasoning",
      "benchmarkKey": "corpusQa1m",
      "name": "CorpusQA 1M",
      "fullName": "CorpusQA 1M",
      "description": "A million-token CorpusQA long-context question-answering benchmark reported in DeepSeek-V4 model evaluations.",
      "paperUrl": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf",
      "paperTitle": "DeepSeek-V4 Technical Report",
      "authors": "DeepSeek-AI",
      "year": "2026",
      "tasks": "Million-token corpus question answering",
      "format": "Long-context QA accuracy",
      "difficulty": "Million-token long context",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/corpusqa1m",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/corpusqa1m.md"
    },
    {
      "category": "reasoning",
      "categoryLabel": "Reasoning",
      "benchmarkKey": "arcAgi2",
      "name": "ARC-AGI-2",
      "fullName": "Abstraction and Reasoning Corpus for AGI v2",
      "description": "A benchmark measuring fluid intelligence and novel abstract reasoning through visual grid puzzles. Models must identify patterns in input-output pairs and generate the correct output for unseen inputs. Considered the hardest public reasoning benchmark — average individual human performance is 66%.",
      "paperUrl": "https://arcprize.org/arc-agi/2/",
      "paperTitle": "ARC-AGI-2: A Harder General Intelligence Benchmark",
      "authors": "Francois Chollet, ARC Prize Foundation",
      "year": 2025,
      "tasks": "Visual pattern completion and abstract reasoning",
      "format": "Grid transformation puzzles with novel rules",
      "difficulty": "Expert-level — hardest public reasoning benchmark",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": 0.31,
      "displayableScoreCount": 21,
      "url": "https://benchlm.ai/benchmarks/arc-agi-2",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/arc-agi-2.md"
    },
    {
      "category": "reasoning",
      "categoryLabel": "Reasoning",
      "benchmarkKey": "arcAgi3",
      "name": "ARC-AGI-3",
      "fullName": "Abstraction and Reasoning Corpus for AGI v3",
      "description": "An interactive successor to ARC-AGI-2 that evaluates whether an AI agent can learn unfamiliar task mechanics through action and feedback.",
      "paperUrl": "https://arcprize.org/media/ARC_AGI_3_Technical_Report.pdf",
      "paperTitle": "ARC-AGI-3: A New Challenge for Frontier Agentic Intelligence",
      "authors": "ARC Prize Foundation",
      "year": 2026,
      "tasks": "Interactive game-like tasks with hidden rules",
      "format": "Agentic task completion under a capped evaluation budget",
      "difficulty": "Frontier agentic reasoning",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 11,
      "url": "https://benchlm.ai/benchmarks/arcagi3",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/arcagi3.md"
    },
    {
      "category": "reasoning",
      "categoryLabel": "Reasoning",
      "benchmarkKey": "geneBenchPro",
      "name": "GeneBench-Pro",
      "fullName": "GeneBench-Pro",
      "description": "A multistage statistical-reasoning benchmark for genomics and biological-data analysis agents.",
      "paperUrl": "https://cdn.openai.com/pdf/21938268-21af-442f-af93-3b2249afb241/genebench-pro.pdf",
      "paperTitle": "GeneBench-Pro: Evaluating Multistage Statistical Reasoning",
      "authors": "OpenAI",
      "year": 2026,
      "tasks": "129 genomics statistical-analysis workflows",
      "format": "Eval-level pass rate across dependent analysis decisions",
      "difficulty": "Long-horizon scientific reasoning",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/genebenchpro",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/genebenchpro.md"
    },
    {
      "category": "reasoning",
      "categoryLabel": "Reasoning",
      "benchmarkKey": "aiNeedle",
      "name": "AI-Needle",
      "fullName": "AI-Needle",
      "description": "A long-context retrieval benchmark that measures whether a model can recover relevant information embedded deep inside very long contexts.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2026",
      "tasks": "Long-context retrieval",
      "format": "Needle-in-a-haystack recall",
      "difficulty": "Long-context memory",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 4,
      "url": "https://benchlm.ai/benchmarks/aineedle",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/aineedle.md"
    },
    {
      "category": "reasoning",
      "categoryLabel": "Reasoning",
      "benchmarkKey": "gpqaDiamond",
      "name": "GPQA Diamond",
      "fullName": "GPQA Diamond",
      "description": "The hardest subset of GPQA featuring the most challenging graduate-level science questions. Sometimes reported separately from the standard GPQA benchmark.",
      "paperUrl": "https://arxiv.org/abs/2311.12022",
      "paperTitle": "GPQA: A Graduate-Level Google-Proof Q&A Benchmark",
      "authors": "David Rein et al.",
      "year": "2023",
      "tasks": "Expert-level science questions",
      "format": "Multiple choice questions",
      "difficulty": "Graduate-level scientific reasoning",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/gpqa-diamond",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/gpqa-diamond.md"
    },
    {
      "category": "reasoning",
      "categoryLabel": "Reasoning",
      "benchmarkKey": "lcr",
      "name": "AA-LCR",
      "fullName": "Artificial Analysis Long Context Reasoning",
      "description": "A display-only Artificial Analysis long-context reasoning evaluation.",
      "paperUrl": "https://artificialanalysis.ai/models/grok-4-3",
      "paperTitle": "Artificial Analysis model benchmarks",
      "authors": "Artificial Analysis",
      "year": "2026",
      "tasks": "Long-context reasoning tasks",
      "format": "Accuracy",
      "difficulty": "Long-context reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 157,
      "url": "https://benchlm.ai/benchmarks/lcr",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/lcr.md"
    },
    {
      "category": "reasoning",
      "categoryLabel": "Reasoning",
      "benchmarkKey": "critpt",
      "name": "CritPt",
      "fullName": "Critical Physics Tasks",
      "description": "A display-only Artificial Analysis metric for research-level physics reasoning.",
      "paperUrl": "https://artificialanalysis.ai/evaluations/critpt",
      "paperTitle": "CritPt Benchmark Leaderboard",
      "authors": "Artificial Analysis",
      "year": "2026",
      "tasks": "Research-level physics questions",
      "format": "Accuracy",
      "difficulty": "Research-level physics reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 160,
      "url": "https://benchlm.ai/benchmarks/critpt",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/critpt.md"
    },
    {
      "category": "reasoning",
      "categoryLabel": "Reasoning",
      "benchmarkKey": "bullshitBenchV2",
      "name": "BullshitBench v2",
      "fullName": "BullshitBench v2",
      "description": "A benchmark that tests whether AI models challenge nonsensical, ill-posed, or logically flawed prompts instead of confidently generating incorrect answers. Measures the critical ability to push back on bad input.",
      "paperUrl": "https://petergpt.github.io/bullshit-benchmark/",
      "paperTitle": "BullshitBench: Measuring whether AI models challenge nonsensical prompts",
      "authors": "Peter Gostev",
      "year": "2025",
      "tasks": "Nonsensical and flawed prompts across multiple domains",
      "format": "Prompt challenge and refusal evaluation",
      "difficulty": "Robustness and critical reasoning",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/bullshitbenchv2",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/bullshitbenchv2.md"
    },
    {
      "category": "reasoning",
      "categoryLabel": "Reasoning",
      "benchmarkKey": "wildBench",
      "name": "WildBench",
      "fullName": "WildBench",
      "description": "An automated evaluation framework using 1,000+ real-world user tasks covering reasoning, planning, coding, and creative writing. Highly correlated with Chatbot Arena human preference rankings.",
      "paperUrl": "https://arxiv.org/abs/2406.04770",
      "paperTitle": "WildBench: Benchmarking Language Models with Challenging Tasks from Real Users in the Wild",
      "authors": "Bill Yuchen Lin et al.",
      "year": "2024",
      "tasks": "1,024 real-world tasks",
      "format": "Real-world task evaluation",
      "difficulty": "Diverse real-world scenarios",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/wildbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/wildbench.md"
    },
    {
      "category": "instructionFollowing",
      "categoryLabel": "Instruction Following",
      "benchmarkKey": "ifeval",
      "name": "IFEval",
      "fullName": "Instruction-Following Eval",
      "description": "A benchmark of 541 prompts built from 25 verifiable instruction types. It tests whether a model follows checkable constraints such as keyword, length, casing, and response-format requirements.",
      "paperUrl": "https://arxiv.org/abs/2311.07911",
      "paperTitle": "Instruction-Following Evaluation for Large Language Models",
      "authors": "Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou",
      "year": "2023",
      "tasks": "541 prompts across 25 instruction types",
      "format": "Constrained generation",
      "difficulty": "Instruction precision",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": 0.35,
      "displayableScoreCount": 26,
      "url": "https://benchlm.ai/benchmarks/ifeval",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/ifeval.md"
    },
    {
      "category": "instructionFollowing",
      "categoryLabel": "Instruction Following",
      "benchmarkKey": "ifBench",
      "name": "IFBench",
      "fullName": "Instruction Following Benchmark",
      "description": "IFBench evaluates precise instruction-following generalization on 58 challenging, verifiable out-of-domain constraints. Unlike IFEval which tests familiar constraint types, IFBench specifically measures how well models follow novel instructions they haven't been optimized for, exposing overfitting to common instruction patterns.",
      "paperUrl": null,
      "paperTitle": null,
      "authors": null,
      "year": 2025,
      "tasks": 58,
      "format": null,
      "difficulty": null,
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": 0.65,
      "displayableScoreCount": 27,
      "url": "https://benchlm.ai/benchmarks/ifbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/ifbench.md"
    },
    {
      "category": "instructionFollowing",
      "categoryLabel": "Instruction Following",
      "benchmarkKey": "aaIfBench",
      "name": "AA-IFBench",
      "fullName": "Artificial Analysis IFBench",
      "description": "A display-only Artificial Analysis IFBench score.",
      "paperUrl": "https://artificialanalysis.ai/evaluations/ifbench",
      "paperTitle": "Artificial Analysis IFBench Benchmark Leaderboard",
      "authors": "Artificial Analysis",
      "year": "2026",
      "tasks": "Verifiable instruction constraints",
      "format": "Constraint satisfaction accuracy",
      "difficulty": "Instruction precision",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 127,
      "url": "https://benchlm.ai/benchmarks/aaifbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/aaifbench.md"
    },
    {
      "category": "instructionFollowing",
      "categoryLabel": "Instruction Following",
      "benchmarkKey": "sobValueAcc",
      "name": "SOB Value Acc",
      "fullName": "Structured Output Benchmark Value Accuracy",
      "description": "A structured-output benchmark from Interfaze measuring whether extracted JSON leaf values exactly match verified ground truth.",
      "paperUrl": "https://interfaze.ai/leaderboards/structured-output-benchmark",
      "paperTitle": "Structured Output Benchmark Leaderboard",
      "authors": "Interfaze",
      "year": "2026",
      "tasks": "Structured output extraction",
      "format": "Value accuracy",
      "difficulty": "Production structured-output reliability",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/sobvalueacc",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/sobvalueacc.md"
    },
    {
      "category": "multilingual",
      "categoryLabel": "Multilingual",
      "benchmarkKey": "gmmlu",
      "name": "GMMLU",
      "fullName": "Global MMLU",
      "description": "MMLU-style knowledge evaluation across 42 high- and low-resource languages.",
      "paperUrl": "https://arxiv.org/abs/2412.03304",
      "paperTitle": "Global MMLU: Understanding and addressing cultural and linguistic biases in multilingual evaluation",
      "authors": "Singh et al.",
      "year": "2024",
      "tasks": "Knowledge questions across 42 languages",
      "format": "Average accuracy",
      "difficulty": "Multilingual knowledge",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/gmmlu",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/gmmlu.md"
    },
    {
      "category": "multilingual",
      "categoryLabel": "Multilingual",
      "benchmarkKey": "milu",
      "name": "MILU",
      "fullName": "Multi-task Indic Language Understanding Benchmark",
      "description": "Culturally grounded knowledge comprehension across ten Indic languages and English.",
      "paperUrl": "https://arxiv.org/abs/2411.02538",
      "paperTitle": "MILU: A Multi-task Indic language understanding benchmark",
      "authors": "Verma et al.",
      "year": "2024",
      "tasks": "Knowledge tasks across 11 languages",
      "format": "Average accuracy",
      "difficulty": "Multilingual Indic knowledge",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/milu",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/milu.md"
    },
    {
      "category": "multilingual",
      "categoryLabel": "Multilingual",
      "benchmarkKey": "aaGlobalMmluLite",
      "name": "AA Global-MMLU-Lite",
      "fullName": "Artificial Analysis Global-MMLU-Lite",
      "description": "An independently evaluated multilingual knowledge result from Artificial Analysis.",
      "paperUrl": "https://artificialanalysis.ai/evaluations/global-mmlu-lite",
      "paperTitle": "Artificial Analysis Global-MMLU-Lite Benchmark Leaderboard",
      "authors": "Artificial Analysis",
      "year": "2026",
      "tasks": "Multilingual knowledge questions",
      "format": "Accuracy",
      "difficulty": "Multilingual professional knowledge",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 5,
      "url": "https://benchlm.ai/benchmarks/aaglobalmmlulite",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/aaglobalmmlulite.md"
    },
    {
      "category": "multilingual",
      "categoryLabel": "Multilingual",
      "benchmarkKey": "mgsm",
      "name": "MGSM",
      "fullName": "Multilingual Grade School Math",
      "description": "A multilingual benchmark that translates 250 grade school math problems from GSM8K into 10 typologically diverse languages: Bengali, German, Spanish, French, Japanese, Russian, Swahili, Telugu, Thai, and Chinese.",
      "paperUrl": "https://arxiv.org/abs/2210.03057",
      "paperTitle": "Language Models are Multilingual Chain-of-Thought Reasoners",
      "authors": "Freda Shi, Mirac Suzgun, Markus Freitag, Xuezhi Wang, Suraj Srivats, Soroush Vosoughi, Hyung Won Chung, Yi Tay, Sebastian Ruder, Denny Zhou, Dipanjan Das, Jason Wei",
      "year": "2022",
      "tasks": "250 problems × 11 languages",
      "format": "Math word problems",
      "difficulty": "Grade school math, multilingual",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/mgsm",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mgsm.md"
    },
    {
      "category": "multilingual",
      "categoryLabel": "Multilingual",
      "benchmarkKey": "mmluProX",
      "name": "MMLU-ProX",
      "fullName": "MMLU-ProX",
      "description": "A multilingual extension of professional-level academic evaluation across many languages.",
      "paperUrl": "https://arxiv.org/abs/2503.10497",
      "paperTitle": "MMLU-ProX: A Multilingual Benchmark for Advanced Large Language Model Evaluation",
      "authors": "MMLU-ProX authors",
      "year": "2025",
      "tasks": "Multilingual professional QA",
      "format": "Multilingual multiple choice",
      "difficulty": "Professional multilingual",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": 1,
      "displayableScoreCount": 12,
      "url": "https://benchlm.ai/benchmarks/mmluprox",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mmluprox.md"
    },
    {
      "category": "multilingual",
      "categoryLabel": "Multilingual",
      "benchmarkKey": "nova63",
      "name": "NOVA-63",
      "fullName": "NOVA-63",
      "description": "A broad multilingual benchmark row from Qwen's launch comparisons intended to measure cross-lingual capability beyond a single language family.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2026",
      "tasks": "Broad multilingual evaluation",
      "format": "Cross-lingual benchmark",
      "difficulty": "Broad multilingual capability",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 7,
      "url": "https://benchlm.ai/benchmarks/nova63",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/nova63.md"
    },
    {
      "category": "multilingual",
      "categoryLabel": "Multilingual",
      "benchmarkKey": "include",
      "name": "INCLUDE",
      "fullName": "INCLUDE",
      "description": "A multilingual benchmark used in provider tables to measure inclusive language coverage and cross-lingual understanding beyond common high-resource languages.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2026",
      "tasks": "Cross-lingual understanding",
      "format": "Multilingual benchmark",
      "difficulty": "Broad multilingual capability",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 4,
      "url": "https://benchlm.ai/benchmarks/include",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/include.md"
    },
    {
      "category": "multilingual",
      "categoryLabel": "Multilingual",
      "benchmarkKey": "polyMath",
      "name": "PolyMath",
      "fullName": "PolyMath",
      "description": "A multilingual mathematical reasoning benchmark that tests whether math performance transfers across languages rather than only in English.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2026",
      "tasks": "Multilingual math problems",
      "format": "Cross-lingual mathematical reasoning",
      "difficulty": "Advanced multilingual reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/polymath",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/polymath.md"
    },
    {
      "category": "multilingual",
      "categoryLabel": "Multilingual",
      "benchmarkKey": "vwt2kLite",
      "name": "VWT2k-lite",
      "fullName": "VWT2k-lite",
      "description": "A lighter multilingual benchmark slice published in provider tables for broad cross-lingual transfer and understanding.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2026",
      "tasks": "Multilingual transfer tasks",
      "format": "Cross-lingual benchmark",
      "difficulty": "Broad multilingual capability",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/vwt2klite",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/vwt2klite.md"
    },
    {
      "category": "multilingual",
      "categoryLabel": "Multilingual",
      "benchmarkKey": "maxife",
      "name": "MAXIFE",
      "fullName": "MAXIFE",
      "description": "A multilingual instruction-following and understanding benchmark row published in Qwen's launch comparisons.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2026",
      "tasks": "Multilingual instruction following",
      "format": "Cross-lingual benchmark",
      "difficulty": "Advanced multilingual instruction following",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/maxife",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/maxife.md"
    },
    {
      "category": "multilingual",
      "categoryLabel": "Multilingual",
      "benchmarkKey": "sweMultilingual",
      "name": "SWE Multilingual",
      "fullName": "SWE-bench Multilingual",
      "description": "A multilingual extension of SWE-bench covering 300 problems across 9 programming languages, testing code generation and bug fixing beyond Python.",
      "paperUrl": "https://www.swebench.com/multilingual",
      "paperTitle": "SWE-bench Multilingual",
      "authors": "SWE-bench team",
      "year": "2025",
      "tasks": "300 problems across 9 languages",
      "format": "Multi-language code patch generation",
      "difficulty": "Professional multilingual software engineering",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/swe-bench-multilingual",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/swe-bench-multilingual.md"
    },
    {
      "category": "multilingual",
      "categoryLabel": "Multilingual",
      "benchmarkKey": "nanoBeirMultilingual",
      "name": "NanoBEIR Multilingual",
      "fullName": "NanoBEIR Multilingual Extended",
      "description": "A display-only multilingual retrieval benchmark reported by Liquid AI for LFM2.5 retriever models, using NDCG@10 across 11 languages.",
      "paperUrl": "https://www.liquid.ai/blog/lfm2-5-retrievers",
      "paperTitle": "LFM2.5 Retrievers: Bi-directional LFMs for Fast Multilingual Search",
      "authors": "Liquid AI",
      "year": "2026",
      "tasks": "Multilingual document retrieval",
      "format": "NDCG@10 average",
      "difficulty": "Multilingual retrieval",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/nanobeirmultilingual",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/nanobeirmultilingual.md"
    },
    {
      "category": "multilingual",
      "categoryLabel": "Multilingual",
      "benchmarkKey": "mkqa11",
      "name": "MKQA-11",
      "fullName": "MKQA-11 multilingual retrieval",
      "description": "A display-only multilingual QA retrieval benchmark reported by Liquid AI for LFM2.5 retriever models, using Recall@20 across 11 languages.",
      "paperUrl": "https://www.liquid.ai/blog/lfm2-5-retrievers",
      "paperTitle": "LFM2.5 Retrievers: Bi-directional LFMs for Fast Multilingual Search",
      "authors": "Liquid AI",
      "year": "2026",
      "tasks": "Cross-lingual open-domain QA retrieval",
      "format": "Recall@20 average",
      "difficulty": "Multilingual retrieval",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/mkqa11",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mkqa11.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "draco",
      "name": "DRACO",
      "fullName": "Data Research and Analysis with Complex Operations",
      "description": "Agentic data-analysis tasks scored against per-task rubrics at a 980K-token budget.",
      "paperUrl": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",
      "paperTitle": "Claude Opus 5 System Card",
      "authors": "Anthropic",
      "year": "2026",
      "tasks": "Agentic data research and analysis tasks",
      "format": "Normalized rubric score",
      "difficulty": "Professional data analysis",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 3,
      "url": "https://benchlm.ai/benchmarks/draco",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/draco.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "multiAgentBrowseCompPrerelease",
      "name": "BrowseComp (10-agent, prerelease)",
      "fullName": "Multi-Agent BrowseComp — 10-agent team prerelease configuration",
      "description": "BrowseComp accuracy from ten collaborating Opus 5 agents on a pre-release model and unreleased effort configuration.",
      "paperUrl": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",
      "paperTitle": "Claude Opus 5 System Card",
      "authors": "Anthropic",
      "year": "2026",
      "tasks": "BrowseComp web-research tasks",
      "format": "10-agent team accuracy",
      "difficulty": "Long-horizon web research",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/multiagentbrowsecompprerelease",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/multiagentbrowsecompprerelease.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "mcpAtlasClaimCoverage",
      "name": "MCP-Atlas claim coverage",
      "fullName": "MCP-Atlas mean claim coverage",
      "description": "Average coverage of required claims in answers produced during real-world MCP tool-use workflows.",
      "paperUrl": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",
      "paperTitle": "Claude Opus 5 System Card",
      "authors": "Anthropic",
      "year": "2026",
      "tasks": "Production-like multi-server MCP workflows",
      "format": "Mean claim coverage",
      "difficulty": "Real-world tool use",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/mcpatlasclaimcoverage",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mcpatlasclaimcoverage.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "legalAgentBenchAllPass",
      "name": "LAB all-pass (Anthropic harness)",
      "fullName": "Legal Agent Benchmark all-pass rate — Anthropic harness",
      "description": "Strict task success requiring every expert-written legal-work rubric criterion to pass.",
      "paperUrl": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",
      "paperTitle": "Claude Opus 5 System Card",
      "authors": "Harvey AI and Anthropic",
      "year": "2026",
      "tasks": "1,235 legal-agent tasks",
      "format": "All-criteria pass rate",
      "difficulty": "Professional legal work",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/legalagentbenchallpass",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/legalagentbenchallpass.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "legalAgentBenchCriterionPass",
      "name": "LAB criterion-pass (Anthropic harness)",
      "fullName": "Legal Agent Benchmark mean criterion-pass rate — Anthropic harness",
      "description": "Mean fraction of expert-written rubric criteria passed across legal-agent tasks.",
      "paperUrl": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",
      "paperTitle": "Claude Opus 5 System Card",
      "authors": "Harvey AI and Anthropic",
      "year": "2026",
      "tasks": "1,235 legal-agent tasks",
      "format": "Mean criterion-pass rate",
      "difficulty": "Professional legal work",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/legalagentbenchcriterionpass",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/legalagentbenchcriterionpass.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "legalAgentBenchHeldoutAllPass",
      "name": "LAB all-pass (Harvey held-out)",
      "fullName": "Legal Agent Benchmark all-pass rate — Harvey held-out set",
      "description": "Harvey AI's strict held-out task success rate requiring every rubric criterion to pass.",
      "paperUrl": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",
      "paperTitle": "Claude Opus 5 System Card",
      "authors": "Harvey AI",
      "year": "2026",
      "tasks": "Harvey-held-out legal-agent tasks",
      "format": "All-criteria pass rate",
      "difficulty": "Professional legal work",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/legalagentbenchheldoutallpass",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/legalagentbenchheldoutallpass.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "legalAgentBenchHeldoutCriterionPass",
      "name": "LAB criterion-pass (Harvey held-out)",
      "fullName": "Legal Agent Benchmark mean criterion-pass rate — Harvey held-out set",
      "description": "Harvey AI's mean criterion-level score on its held-out legal-agent evaluation.",
      "paperUrl": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",
      "paperTitle": "Claude Opus 5 System Card",
      "authors": "Harvey AI",
      "year": "2026",
      "tasks": "Harvey-held-out legal-agent tasks",
      "format": "Mean criterion-pass rate",
      "difficulty": "Professional legal work",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/legalagentbenchheldoutcriterionpass",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/legalagentbenchheldoutcriterionpass.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "toolathlonVerifiedPass3",
      "name": "Toolathlon Verified Pass@3",
      "fullName": "Toolathlon Verified Pass@3",
      "description": "Fraction of Toolathlon Verified tasks solved in at least one of three trials.",
      "paperUrl": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",
      "paperTitle": "Claude Opus 5 System Card",
      "authors": "Anthropic",
      "year": "2026",
      "tasks": "108 verified real-world tool-use tasks",
      "format": "Pass@3",
      "difficulty": "Long-horizon application tool use",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/toolathlonverifiedpass3",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/toolathlonverifiedpass3.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "toolathlonVerifiedPass3All",
      "name": "Toolathlon Verified Pass³",
      "fullName": "Toolathlon Verified Pass cubed",
      "description": "Fraction of Toolathlon Verified tasks solved in all three independent trials.",
      "paperUrl": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",
      "paperTitle": "Claude Opus 5 System Card",
      "authors": "Anthropic",
      "year": "2026",
      "tasks": "108 verified real-world tool-use tasks",
      "format": "All-three-trials pass rate",
      "difficulty": "Long-horizon application tool use",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/toolathlonverifiedpass3all",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/toolathlonverifiedpass3all.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "toolathlonVerifiedAvgTurns",
      "name": "Toolathlon Verified avg. turns",
      "fullName": "Toolathlon Verified average assistant turns",
      "description": "Average assistant turns per Toolathlon Verified trajectory.",
      "paperUrl": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",
      "paperTitle": "Claude Opus 5 System Card",
      "authors": "Anthropic",
      "year": "2026",
      "tasks": "108 verified real-world tool-use tasks",
      "format": "Average trajectory length",
      "difficulty": "Long-horizon application tool use",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/toolathlonverifiedavgturns",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/toolathlonverifiedavgturns.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "frontierBench",
      "name": "Terminal-Bench 3.0",
      "fullName": "Terminal-Bench 3.0",
      "description": "A continuously maintained benchmark for difficult computer work, including coding, deep learning, finance, engineering, math, and science tasks.",
      "paperUrl": "https://www.frontierbench.ai/",
      "paperTitle": "Terminal-Bench 3.0",
      "authors": "Ryan Marten, Alex Shaw, Andy Konwinski, Harbor, and the Laude Institute",
      "year": "2026",
      "tasks": "74 professional computer-work tasks across 7 domains",
      "format": "Task completion rate",
      "difficulty": "Frontier autonomous knowledge work",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": "https://benchlm.ai/benchmarks/terminal-bench-4",
      "weight": null,
      "displayableScoreCount": 10,
      "url": "https://benchlm.ai/benchmarks/terminal-bench-3",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/terminal-bench-3.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "terminalBench4",
      "name": "Terminal-Bench 4.0",
      "fullName": "Terminal-Bench 4.0",
      "description": "The current Terminal-Bench release measures difficult computer work after recalibrating task resources, fixing unstable tasks, and removing tasks that no longer separate frontier systems.",
      "paperUrl": "https://www.tbench.ai/news/terminal-bench-4-0",
      "paperTitle": "Terminal-Bench 4.0",
      "authors": "Ryan Marten, Terminal-Bench contributors, and Harbor",
      "year": "2026",
      "tasks": "66 professional computer-work tasks",
      "format": "Task completion rate across 5 trials per task",
      "difficulty": "Frontier autonomous knowledge work",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/terminal-bench-4",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/terminal-bench-4.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "terminalBenchScience",
      "name": "Terminal-Bench-Science 0.1",
      "fullName": "Terminal-Bench-Science 0.1",
      "description": "A benchmark of AI agents completing expert-curated research workflows across the life, physical, Earth, mathematical, and engineering sciences.",
      "paperUrl": "https://www.terminal-bench-science.ai/",
      "paperTitle": "Terminal-Bench-Science 0.1",
      "authors": "Terminal-Bench-Science Team",
      "year": "2026",
      "tasks": "70 expert-curated scientific research workflows",
      "format": "Resolution rate across 3 trials per task",
      "difficulty": "Frontier scientific research workflows",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/terminal-bench-science",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/terminal-bench-science.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "eqBench4",
      "name": "EQ-Bench 4",
      "fullName": "EQ-Bench 4",
      "description": "A multi-turn benchmark of emotional and social intelligence using synthetic personas and pairwise LLM judging.",
      "paperUrl": "https://eqbench.com/",
      "paperTitle": "EQ-Bench 4",
      "authors": "EQ-Bench",
      "year": "2026",
      "tasks": "120 multi-turn persona scenarios",
      "format": "Pairwise Elo",
      "difficulty": "Emotional and social intelligence",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/eqbench4",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/eqbench4.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "autoCadBench",
      "name": "AutoCAD-Bench",
      "fullName": "AutoCAD-Bench",
      "description": "A Markov Studios computer-use benchmark that asks agents to produce 2D drawings and 3D models in AutoCAD.",
      "paperUrl": "https://www.markovstudios.com/research/autocad-bench",
      "paperTitle": "AutoCAD-Bench",
      "authors": "Markov Studios",
      "year": "2026",
      "tasks": "21 2D drawing tasks and 29 3D modeling tasks",
      "format": "Task completion rate at a 75-point rubric threshold",
      "difficulty": "Professional CAD computer use",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/autocadbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/autocadbench.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "designArenaAgenticWebDev",
      "name": "Design Arena Agentic Web Dev",
      "fullName": "Design Arena Agentic Web Dev Elo",
      "description": "A display-only Elo rating from blinded comparisons of multi-file web applications built by coding agents.",
      "paperUrl": "https://intelligence.ai/leaderboard/webapps",
      "paperTitle": "Design Arena Web Dev (Agentic) Leaderboard",
      "authors": "Design Arena / Intelligence",
      "year": "2026",
      "tasks": "Multi-file web application development",
      "format": "Elo from blinded human preferences",
      "difficulty": "Agentic frontend development",
      "decimals": 0,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/designarenaagenticwebdev",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/designarenaagenticwebdev.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "aaBriefcaseElo",
      "name": "AA Briefcase",
      "fullName": "Artificial Analysis Briefcase",
      "description": "An independently evaluated professional-work benchmark reported as Elo.",
      "paperUrl": "https://artificialanalysis.ai/evaluations/aa-briefcase",
      "paperTitle": "Artificial Analysis Briefcase Benchmark Leaderboard",
      "authors": "Artificial Analysis",
      "year": "2026",
      "tasks": "Professional knowledge-work tasks",
      "format": "Elo",
      "difficulty": "Professional work",
      "decimals": 0,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 16,
      "url": "https://benchlm.ai/benchmarks/aabriefcaseelo",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/aabriefcaseelo.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "aaAutomationBench",
      "name": "AA AutomationBench",
      "fullName": "Artificial Analysis AutomationBench",
      "description": "An independently evaluated automation benchmark from Artificial Analysis.",
      "paperUrl": "https://artificialanalysis.ai/evaluations/automationbench-aa",
      "paperTitle": "Artificial Analysis AutomationBench Benchmark Leaderboard",
      "authors": "Artificial Analysis",
      "year": "2026",
      "tasks": "Business-process automation tasks",
      "format": "Task success rate",
      "difficulty": "Agentic automation",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 12,
      "url": "https://benchlm.ai/benchmarks/aaautomationbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/aaautomationbench.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "aaEnterpriseOpsGym",
      "name": "AA EnterpriseOps-Gym",
      "fullName": "Artificial Analysis EnterpriseOps-Gym",
      "description": "An independently evaluated enterprise-operations benchmark from Artificial Analysis.",
      "paperUrl": "https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa",
      "paperTitle": "Artificial Analysis EnterpriseOps-Gym Benchmark Leaderboard",
      "authors": "Artificial Analysis",
      "year": "2026",
      "tasks": "Enterprise operations workflows",
      "format": "Task success rate",
      "difficulty": "Enterprise agent operations",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 17,
      "url": "https://benchlm.ai/benchmarks/aaenterpriseopsgym",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/aaenterpriseopsgym.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "aaHarveyLab",
      "name": "AA Harvey LAB",
      "fullName": "Artificial Analysis Harvey LAB-AA",
      "description": "An independently evaluated legal-agent benchmark from Artificial Analysis.",
      "paperUrl": "https://artificialanalysis.ai/evaluations/harvey-lab-aa",
      "paperTitle": "Artificial Analysis Harvey LAB-AA Benchmark Leaderboard",
      "authors": "Artificial Analysis",
      "year": "2026",
      "tasks": "Legal agent tasks",
      "format": "Task success rate",
      "difficulty": "Professional legal work",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 12,
      "url": "https://benchlm.ai/benchmarks/aaharveylab",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/aaharveylab.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "aaItbench",
      "name": "AA ITBench",
      "fullName": "Artificial Analysis ITBench-AA",
      "description": "An independently evaluated IT-operations benchmark from Artificial Analysis.",
      "paperUrl": "https://artificialanalysis.ai/evaluations/itbench-aa",
      "paperTitle": "Artificial Analysis ITBench-AA Benchmark Leaderboard",
      "authors": "Artificial Analysis",
      "year": "2026",
      "tasks": "IT incident-response tasks",
      "format": "Task success rate",
      "difficulty": "Enterprise IT operations",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 7,
      "url": "https://benchlm.ai/benchmarks/aaitbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/aaitbench.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "aaTau3Banking",
      "name": "AA Tau3 Banking",
      "fullName": "Artificial Analysis Tau3-Banking",
      "description": "An independently evaluated Tau3 banking benchmark from Artificial Analysis.",
      "paperUrl": "https://artificialanalysis.ai/evaluations/tau3-banking",
      "paperTitle": "Artificial Analysis Tau3-Banking Benchmark Leaderboard",
      "authors": "Artificial Analysis",
      "year": "2026",
      "tasks": "Banking tool-use workflows",
      "format": "Task success rate",
      "difficulty": "Agentic banking workflows",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 19,
      "url": "https://benchlm.ai/benchmarks/aatau3banking",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/aatau3banking.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "terminalBench2",
      "name": "Terminal-Bench 2.0",
      "fullName": "Terminal-Bench 2.0",
      "description": "A benchmark for agentic software engineering tasks executed in real terminal environments. Models must inspect files, run commands, edit code, and recover from errors over multi-step workflows.",
      "paperUrl": "https://www.tbench.ai/",
      "paperTitle": "Terminal-Bench 2.0",
      "authors": "Terminal-Bench contributors",
      "year": "2026",
      "tasks": "Terminal-based software tasks",
      "format": "Interactive CLI agent evaluation",
      "difficulty": "Professional software engineering",
      "decimals": null,
      "successorKey": null,
      "successorUrl": "https://benchlm.ai/benchmarks/terminal-bench-4",
      "weight": 0.38,
      "displayableScoreCount": 65,
      "url": "https://benchlm.ai/benchmarks/terminal-bench-2",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/terminal-bench-2.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "terminalBench21",
      "name": "Terminal-Bench 2.1",
      "fullName": "Terminal-Bench 2.1 (provider run)",
      "description": "A provider-run Terminal-Bench 2.1 result stored separately from the repository's Terminal-Bench 2.0 lane.",
      "paperUrl": "https://api-docs.deepseek.com/zh-cn/updates/",
      "paperTitle": "DeepSeek V4 Flash 0731 update",
      "authors": "DeepSeek-AI",
      "year": "2026",
      "tasks": "Terminal-based software-agent tasks",
      "format": "Interactive task success rate",
      "difficulty": "Professional software engineering",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": "https://benchlm.ai/benchmarks/terminal-bench-4",
      "weight": null,
      "displayableScoreCount": 16,
      "url": "https://benchlm.ai/benchmarks/terminalbench21",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/terminalbench21.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "browseComp",
      "name": "BrowseComp",
      "fullName": "BrowseComp",
      "description": "A benchmark for web-browsing agents that must search, inspect sources, gather evidence, and return the correct answer to research-oriented questions.",
      "paperUrl": "https://openai.com/index/browsecomp/",
      "paperTitle": "BrowseComp",
      "authors": "OpenAI",
      "year": "2025",
      "tasks": "Research questions requiring browsing",
      "format": "Web search and evidence synthesis",
      "difficulty": "Hard web research",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": 0.28,
      "displayableScoreCount": 38,
      "url": "https://benchlm.ai/benchmarks/browsecomp",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/browsecomp.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "hleWithTools",
      "name": "HLE w/ tools",
      "fullName": "Humanity's Last Exam with tools",
      "description": "Tool-augmented Humanity's Last Exam scores reported in DeepSeek-V4 thinking-mode evaluations.",
      "paperUrl": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf",
      "paperTitle": "DeepSeek-V4 Technical Report",
      "authors": "DeepSeek-AI",
      "year": "2026",
      "tasks": "Expert questions with tool use",
      "format": "Pass@1",
      "difficulty": "Frontier tool-augmented reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 17,
      "url": "https://benchlm.ai/benchmarks/hlewithtools",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/hlewithtools.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "gdpvalAa",
      "name": "GDPval-AA",
      "fullName": "GDPval-AA",
      "description": "An agentic real-world work-task evaluation reported as an Elo score in DeepSeek-V4 thinking-mode evaluations.",
      "paperUrl": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/DeepSeek_V4.pdf",
      "paperTitle": "DeepSeek-V4 Technical Report",
      "authors": "DeepSeek-AI",
      "year": "2026",
      "tasks": "Agentic real-world work tasks",
      "format": "Elo",
      "difficulty": "Professional agentic workflows",
      "decimals": 0,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 98,
      "url": "https://benchlm.ai/benchmarks/gdpvalaa",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/gdpvalaa.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "gdpvalAaNormalized",
      "name": "GDPval-AA",
      "fullName": "GDPval-AA normalized",
      "description": "A display-only Artificial Analysis normalized score for economically valuable tasks.",
      "paperUrl": "https://artificialanalysis.ai/models/grok-4-3",
      "paperTitle": "Artificial Analysis model benchmarks",
      "authors": "Artificial Analysis",
      "year": "2026",
      "tasks": "Economically valuable tasks",
      "format": "Normalized score",
      "difficulty": "Professional agentic workflows",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 96,
      "url": "https://benchlm.ai/benchmarks/gdpvalaanormalized",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/gdpvalaanormalized.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "aaAgenticIndex",
      "name": "AA Agentic Index",
      "fullName": "Artificial Analysis Agentic Index",
      "description": "A display-only Artificial Analysis agentic index.",
      "paperUrl": "https://artificialanalysis.ai/leaderboards/models",
      "paperTitle": "Artificial Analysis model leaderboards",
      "authors": "Artificial Analysis",
      "year": "2026",
      "tasks": "Cross-benchmark agentic index",
      "format": "Aggregated model score",
      "difficulty": "Display-only external reference",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 88,
      "url": "https://benchlm.ai/benchmarks/aaagenticindex",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/aaagenticindex.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "apexAgentsAa",
      "name": "APEX-Agents-AA",
      "fullName": "APEX-Agents-AA",
      "description": "Artificial Analysis' implementation of the APEX-Agents benchmark for long-horizon professional-services agent tasks.",
      "paperUrl": "https://artificialanalysis.ai/evaluations/apex-agents-aa",
      "paperTitle": "APEX-Agents-AA Benchmark Leaderboard",
      "authors": "Artificial Analysis / Mercor",
      "year": "2026",
      "tasks": "452 professional-services agent tasks",
      "format": "Pass@1",
      "difficulty": "Long-horizon workplace agent tasks",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 28,
      "url": "https://benchlm.ai/benchmarks/apexagentsaa",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/apexagentsaa.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "gertLabs",
      "name": "Gert Labs",
      "fullName": "Gert Labs Composite Game Benchmark",
      "description": "A game-environment benchmark that evaluates AI models in novel games covering strategic planning, resource management, spatial reasoning, cooperation, and theory of mind.",
      "paperUrl": "https://gertlabs.com/rankings",
      "paperTitle": "Gert Labs rankings",
      "authors": "Gert Labs",
      "year": "2026",
      "tasks": "Novel game environments",
      "format": "Composite game leaderboard",
      "difficulty": "Agentic coding and decision-making",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 52,
      "url": "https://benchlm.ai/benchmarks/gertlabs",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/gertlabs.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "osWorldVerified",
      "name": "OSWorld-Verified",
      "fullName": "OSWorld-Verified",
      "description": "OSWorld-Verified is the July 2025 repaired release of OSWorld's real-computer evaluation. It measures whether a model-agent system can finish desktop and web tasks from configured starting states, with success checked by execution-based evaluators.",
      "paperUrl": "https://os-world.github.io/",
      "paperTitle": "OSWorld",
      "authors": "Tianbao Xie, Danyang Zhang, Jixuan Chen, Xiaochuan Li, Siheng Zhao, Ruisheng Cao, Toh Jing Hua, Zhoujun Cheng, Dongchan Shin, Fangyu Lei, Yitao Liu, Yiheng Xu, Shuyan Zhou, Silvio Savarese, Caiming Xiong, Victor Zhong, Tao Yu",
      "year": "2025",
      "tasks": "369 real-world computer tasks (361 when eight Google Drive tasks are excluded)",
      "format": "Execution-based interactive task success",
      "difficulty": "Multi-step desktop and cross-application workflows",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": 0.34,
      "displayableScoreCount": 30,
      "url": "https://benchlm.ai/benchmarks/osworld-verified",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/osworld-verified.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "osWorld2",
      "name": "OSWorld 2.0",
      "fullName": "OSWorld 2.0",
      "description": "A long-horizon computer-use benchmark covering realistic workflows across everyday and professional desktop tasks.",
      "paperUrl": "https://arxiv.org/abs/2606.29537",
      "paperTitle": "OSWorld2.0: Benchmarking Computer Use Agents on Long-Horizon Real-World Tasks",
      "authors": "Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu",
      "year": "2026",
      "tasks": "108 long-horizon computer-use workflows",
      "format": "Interactive computer-use evaluation",
      "difficulty": "Long-horizon professional workflows",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 16,
      "url": "https://benchlm.ai/benchmarks/osworld2",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/osworld2.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "cyberGym",
      "name": "CyberGym",
      "fullName": "CyberGym",
      "description": "A cybersecurity task benchmark for evaluating defensive cyber workflows and vulnerability-oriented agent performance.",
      "paperUrl": "https://www.cybergym.io/",
      "paperTitle": "CyberGym: Evaluating AI Agents' Real-World Cybersecurity Capabilities at Scale",
      "authors": "Zhun Wang, Tianneng Shi, Jingxuan He, Matthew Cai, Jialin Zhang, Dawn Song",
      "year": "2026",
      "tasks": "1,507 vulnerability analysis instances",
      "format": "Vulnerability reproduction and PoC generation",
      "difficulty": "Real-world cybersecurity",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 21,
      "url": "https://benchlm.ai/benchmarks/cybergym",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/cybergym.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "ctiRealm",
      "name": "CTI-REALM",
      "fullName": "CTI-REALM",
      "description": "A cybersecurity benchmark that measures whether an agent can turn raw threat-intelligence reports into working detection rules.",
      "paperUrl": "https://sakana.ai/fugu-cyber-release/",
      "paperTitle": "Introducing Fugu-Cyber",
      "authors": "Sakana AI",
      "year": "2026",
      "tasks": "Threat-intelligence-to-detection-rule workflows",
      "format": "Success rate",
      "difficulty": "Professional cyber threat detection",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/ctirealm",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/ctirealm.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "cybench",
      "name": "Cybench",
      "fullName": "Cybench",
      "description": "A cybersecurity benchmark of professional Capture the Flag tasks for measuring autonomous cyber agent capability and risk.",
      "paperUrl": "https://arxiv.org/abs/2408.08926",
      "paperTitle": "Cybench: A Framework for Evaluating Cybersecurity Capabilities and Risks of Language Models",
      "authors": "Andy K. Zhang, Neil Perry, Riya Dulepet, Joey Ji, Celeste Menders, Justin W. Lin, Eliot Jones, Gashon Hussein, Samantha Liu, Donovan Jasper, Pura Peetathawatchai, Ari Glenn, Vikram Sivashankar, Daniel Zamoshchin, Leo Glikbarg, Derek Askaryar, Mike Yang, Teddy Zhang, Rishi Alluri, Nathan Tran, Rinnara Sangpisit, Polycarpos Yiorkadjis, Kenny Osele, Gautham Raghupathi, Dan Boneh, Daniel E. Ho, Percy Liang",
      "year": "2025",
      "tasks": "40 professional CTF tasks",
      "format": "Cybersecurity agent task completion",
      "difficulty": "Professional cybersecurity",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/cybench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/cybench.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "exploitGym",
      "name": "ExploitGym",
      "fullName": "ExploitGym",
      "description": "A controlled benchmark for evaluating whether AI agents can extend vulnerability-triggering inputs into working exploits.",
      "paperUrl": "https://arxiv.org/abs/2605.11086",
      "paperTitle": "ExploitGym: Can AI Agents Turn Security Vulnerabilities into Real Attacks?",
      "authors": "Zhun Wang, Nico Schiller, Hongwei Li, Srijiith Sesha Narayana, Milad Nasr, Nicholas Carlini, Xiangyu Qi, Eric Wallace, Elie Bursztein, Luca Invernizzi, Kurt Thomas, Yan Shoshitaishvili, Wenbo Guo, Jingxuan He, Thorsten Holz, Dawn Song",
      "year": "2026",
      "tasks": "898 exploitation tasks",
      "format": "Working exploit generation",
      "difficulty": "Advanced cybersecurity exploitation",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 8,
      "url": "https://benchlm.ai/benchmarks/exploitgym",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/exploitgym.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "jobBench",
      "name": "JobBench",
      "fullName": "JobBench",
      "description": "An occupational agent benchmark for professional workflows that workers say they most want delegated to AI.",
      "paperUrl": "https://arxiv.org/abs/2605.26329",
      "paperTitle": "JobBench: Aligning Agent Work With Human Will",
      "authors": "Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran",
      "year": "2026",
      "tasks": "130 tasks across 35 occupations",
      "format": "Agentic workplace deliverables",
      "difficulty": "Professional multi-source workflows",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 25,
      "url": "https://benchlm.ai/benchmarks/jobbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/jobbench.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "browseCompVl",
      "name": "BrowseComp-VL",
      "fullName": "BrowseComp-VL",
      "description": "A vision-language browsing benchmark for multimodal web research and tool-use workflows.",
      "paperUrl": "https://docs.z.ai/guides/vlm/glm-5v-turbo",
      "paperTitle": "GLM-5V-Turbo",
      "authors": "Z.AI",
      "year": "2026",
      "tasks": "Multimodal browsing tasks",
      "format": "Vision-language web research evaluation",
      "difficulty": "Multimodal browser-agent",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/browsecompvl",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/browsecompvl.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "osWorld",
      "name": "OSWorld",
      "fullName": "OSWorld",
      "description": "A computer-use benchmark for GUI task completion across the broader OSWorld task suite.",
      "paperUrl": "https://docs.z.ai/guides/vlm/glm-5v-turbo",
      "paperTitle": "GLM-5V-Turbo",
      "authors": "Z.AI",
      "year": "2026",
      "tasks": "Computer-use tasks",
      "format": "Interactive GUI evaluation",
      "difficulty": "Broad computer-use suite",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/osworld",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/osworld.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "androidWorld",
      "name": "AndroidWorld",
      "fullName": "AndroidWorld",
      "description": "A mobile GUI agent benchmark for completing Android app workflows and on-device tasks.",
      "paperUrl": "https://docs.z.ai/guides/vlm/glm-5v-turbo",
      "paperTitle": "GLM-5V-Turbo",
      "authors": "Z.AI",
      "year": "2026",
      "tasks": "Android app workflows",
      "format": "Interactive mobile-agent evaluation",
      "difficulty": "Complex mobile task completion",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 8,
      "url": "https://benchlm.ai/benchmarks/androidworld",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/androidworld.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "webVoyager",
      "name": "WebVoyager",
      "fullName": "WebVoyager",
      "description": "A browser-agent benchmark for completing multi-step workflows on live websites.",
      "paperUrl": "https://docs.z.ai/guides/vlm/glm-5v-turbo",
      "paperTitle": "GLM-5V-Turbo",
      "authors": "Z.AI",
      "year": "2026",
      "tasks": "Live website workflows",
      "format": "Interactive browser-agent evaluation",
      "difficulty": "Multi-step web navigation",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/webvoyager",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/webvoyager.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "mcpAtlas",
      "name": "MCP Atlas",
      "fullName": "MCP Atlas",
      "description": "A benchmark for tool-calling over Model Context Protocol integrations and external tools.",
      "paperUrl": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
      "paperTitle": "Introducing GPT-5.4 mini and nano",
      "authors": "OpenAI",
      "year": "2026",
      "tasks": "Tool-integrated agent tasks",
      "format": "Interactive tool-calling evaluation",
      "difficulty": "Advanced tool use",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 33,
      "url": "https://benchlm.ai/benchmarks/mcpatlas",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mcpatlas.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "kimiClaw247",
      "name": "Kimi Claw 24/7",
      "fullName": "Kimi Claw 24/7 Bench",
      "description": "A Moonshot AI internal long-horizon agent benchmark for persistent professional coworking tasks.",
      "paperUrl": "https://huggingface.co/moonshotai/Kimi-K2.7-Code",
      "paperTitle": "Kimi K2.7 Code",
      "authors": "Moonshot AI",
      "year": "2026",
      "tasks": "17 professional scenarios, 610 evaluation points",
      "format": "Average pass rate across repeated OpenClaw runs",
      "difficulty": "Long-horizon agentic work",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/kimiclaw247",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/kimiclaw247.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "mcpMarkVerified",
      "name": "MCP Mark Verified",
      "fullName": "MCPMark-Verified",
      "description": "A human-verified edition of MCPMark for MCP tool use across Notion, GitHub, Filesystem, Postgres, and Playwright server environments.",
      "paperUrl": "https://mcpmark.ai/",
      "paperTitle": "MCPMark",
      "authors": "MCPMark",
      "year": "2026",
      "tasks": "MCP tool-use tasks across five server environments",
      "format": "Interactive MCP task completion",
      "difficulty": "Advanced tool use",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/mcpmarkverified",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mcpmarkverified.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "toolathlon",
      "name": "Toolathlon",
      "fullName": "Toolathlon",
      "description": "A tool-use benchmark focused on selecting, sequencing, and completing tasks with external tools.",
      "paperUrl": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
      "paperTitle": "Introducing GPT-5.4 mini and nano",
      "authors": "OpenAI",
      "year": "2026",
      "tasks": "Multi-tool workflows",
      "format": "Interactive tool-calling evaluation",
      "difficulty": "Advanced tool use",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 22,
      "url": "https://benchlm.ai/benchmarks/toolathlon",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/toolathlon.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "toolathlonVerified",
      "name": "Toolathlon-Verified",
      "fullName": "Toolathlon-Verified",
      "description": "A verified tool-use benchmark variant for completing multi-step workflows with external tools.",
      "paperUrl": "https://www.kimi.com/blog/kimi-k3",
      "paperTitle": "Kimi K3: Open Frontier Intelligence",
      "authors": "Moonshot AI",
      "year": "2026",
      "tasks": "Verified multi-tool workflows",
      "format": "Interactive tool-use score",
      "difficulty": "Advanced tool use",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 16,
      "url": "https://benchlm.ai/benchmarks/toolathlonverified",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/toolathlonverified.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "automationBench",
      "name": "AutomationBench",
      "fullName": "AutomationBench",
      "description": "An agent benchmark for completing automation workflows in reproducible task environments.",
      "paperUrl": "https://www.kimi.com/blog/kimi-k3",
      "paperTitle": "Kimi K3: Open Frontier Intelligence",
      "authors": "Moonshot AI",
      "year": "2026",
      "tasks": "600 public automation tasks",
      "format": "Agent task-completion score",
      "difficulty": "Long-horizon automation",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 9,
      "url": "https://benchlm.ai/benchmarks/automationbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/automationbench.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "agentsLastExam",
      "name": "Agents' Last Exam",
      "fullName": "Agents' Last Exam",
      "description": "An agent benchmark reported in DeepSeek's V4 Flash 0731 launch comparison.",
      "paperUrl": "https://api-docs.deepseek.com/zh-cn/updates/",
      "paperTitle": "DeepSeek V4 Flash 0731 update",
      "authors": "DeepSeek-AI",
      "year": "2026",
      "tasks": "Agent tasks",
      "format": "Provider-reported task score",
      "difficulty": "Advanced agentic work",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 8,
      "url": "https://benchlm.ai/benchmarks/agentslastexam",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/agentslastexam.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "apexAgents",
      "name": "APEX-Agents",
      "fullName": "APEX-Agents",
      "description": "A professional-services agent benchmark covering long-horizon knowledge-work tasks.",
      "paperUrl": "https://www.kimi.com/blog/kimi-k3",
      "paperTitle": "Kimi K3: Open Frontier Intelligence",
      "authors": "Moonshot AI / APEX-Agents benchmark authors",
      "year": "2026",
      "tasks": "Professional-services agent tasks",
      "format": "Agent task-completion score",
      "difficulty": "Long-horizon professional work",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 7,
      "url": "https://benchlm.ai/benchmarks/apexagents",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/apexagents.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "spreadsheetBench2",
      "name": "SpreadsheetBench 2",
      "fullName": "SpreadsheetBench 2",
      "description": "A spreadsheet-focused benchmark for agentic analysis and editing workflows.",
      "paperUrl": "https://www.kimi.com/blog/kimi-k3",
      "paperTitle": "Kimi K3: Open Frontier Intelligence",
      "authors": "Moonshot AI",
      "year": "2026",
      "tasks": "Spreadsheet analysis and editing tasks",
      "format": "Agent task-completion score",
      "difficulty": "Professional spreadsheet work",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/spreadsheetbench2",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/spreadsheetbench2.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "deckBench",
      "name": "DECK-Bench",
      "fullName": "DECK-Bench (Internal)",
      "description": "Moonshot AI's internal benchmark for presentation and deck-production workflows.",
      "paperUrl": "https://www.kimi.com/blog/kimi-k3",
      "paperTitle": "Kimi K3: Open Frontier Intelligence",
      "authors": "Moonshot AI",
      "year": "2026",
      "tasks": "Internal presentation workflows",
      "format": "Internal evaluation score",
      "difficulty": "Professional presentation creation",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/deckbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/deckbench.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "zClawBench",
      "name": "ZClawBench",
      "fullName": "ZClawBench",
      "description": "A Z.AI benchmark for OpenClaw-style agent workflows spanning information search, office work, data analysis, development and operations, automation, and security.",
      "paperUrl": "https://docs.z.ai/guides/llm/glm-5-turbo",
      "paperTitle": "GLM-5-Turbo",
      "authors": "Z.AI",
      "year": "2026",
      "tasks": "OpenClaw agent workflows",
      "format": "End-to-end agent benchmark",
      "difficulty": "Broad productivity and operations workflows",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/zclawbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/zclawbench.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "tau2Bench",
      "name": "τ²-bench results",
      "fullName": "τ²-Bench Tool-Agent-User Evaluation",
      "description": "This route is a sourced ledger for published τ²-bench results. Most current rows come from Artificial Analysis's telecom implementation, while named provider rows can use telecom, airline, retail, or aggregate setups.",
      "paperUrl": "https://arxiv.org/abs/2506.07982",
      "paperTitle": "τ²-Bench: Evaluating Conversational Agents in a Dual-Control Environment",
      "authors": "Victor Barres, Honghua Dong, Soham Ray, Xujie Si, Karthik Narasimhan",
      "year": "2025",
      "tasks": "Airline, retail, and telecom customer-service task sets",
      "format": "Published domain success or pass^k results",
      "difficulty": "Dual-control customer-service workflows",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 126,
      "url": "https://benchlm.ai/benchmarks/tau2-bench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/tau2-bench.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "deepSearchQa",
      "name": "DeepSearchQA",
      "fullName": "DeepSearchQA",
      "description": "An agentic browsing benchmark where models search the web, gather evidence, and answer list-style questions using browser tools.",
      "paperUrl": "https://ai.meta.com/static-resource/muse-spark-eval-methodology",
      "paperTitle": "Muse Spark Eval Methodology",
      "authors": "Meta AI",
      "year": "2026",
      "tasks": "Agentic browsing and list-answer questions",
      "format": "Search / open / find browser-agent evaluation",
      "difficulty": "Agentic web research",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 15,
      "url": "https://benchlm.ai/benchmarks/deepsearchqa",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/deepsearchqa.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "tau2Airline",
      "name": "τ²-bench Airline",
      "fullName": "τ²-Bench Airline Domain",
      "description": "τ²-bench Airline tests conversational agents on airline customer-service tasks governed by domain policy and database-changing tools.",
      "paperUrl": "https://arxiv.org/abs/2506.07982",
      "paperTitle": "τ²-Bench: Evaluating Conversational Agents in a Dual-Control Environment",
      "authors": "Victor Barres, Honghua Dong, Soham Ray, Xujie Si, Karthik Narasimhan",
      "year": "2025",
      "tasks": "Airline customer-service tasks",
      "format": "Domain success under a published trial policy",
      "difficulty": "Policy-constrained airline support workflows",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/tau2airline",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/tau2airline.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "pinchBench",
      "name": "PinchBench",
      "fullName": "PinchBench",
      "description": "An OpenClaw agent benchmark from Kilo that measures successful task completion across standardized real-world agent workflows.",
      "paperUrl": "https://pinchbench.com/about",
      "paperTitle": "About PinchBench",
      "authors": "Kilo Code",
      "year": "2026",
      "tasks": "23 OpenClaw agent tasks",
      "format": "Average success rate from official runs",
      "difficulty": "Long-horizon agent workflows",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 4,
      "url": "https://benchlm.ai/benchmarks/pinchbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/pinchbench.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "openHandsIndex",
      "name": "OpenHands Index",
      "fullName": "OpenHands Index",
      "description": "A holistic coding-agent benchmark that evaluates AI agents across issue resolution, frontend work, greenfield development, testing, and information gathering.",
      "paperUrl": "https://index.openhands.dev/about",
      "paperTitle": "OpenHands Index methodology",
      "authors": "OpenHands",
      "year": "2025",
      "tasks": "SWE-bench Verified, SWE-bench Multimodal, Commit0, SWT-bench Verified, and GAIA",
      "format": "Macro-average across five coding-agent categories",
      "difficulty": "Real-world software engineering agent tasks",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/openhandsindex",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/openhandsindex.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "sweAtlasRefactoring",
      "name": "SWE-Atlas Refactoring",
      "fullName": "SWE-Atlas Refactoring",
      "description": "A Scale SWE-Atlas software-engineering agent benchmark focused on refactoring tasks.",
      "paperUrl": "https://labs.scale.com/papers/sweatlas",
      "paperTitle": "SWE-Atlas",
      "authors": "Scale AI",
      "year": "2026",
      "tasks": "SWE-Atlas refactoring tasks",
      "format": "Refactoring score with confidence intervals",
      "difficulty": "Real-world software-engineering agent tasks",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/sweatlasrefactoring",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/sweatlasrefactoring.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "sweRefactorBench",
      "name": "SWE Refactor Bench",
      "fullName": "SWE Refactor Bench: Can Coding Agents Complete a Long-Horizon, Whole-Repository Stack Migration?",
      "description": "Tests whether coding agents can complete long-horizon, whole-repository stack migrations while preserving the original program's behavior.",
      "paperUrl": "https://arxiv.org/abs/2608.23564",
      "paperTitle": "SWE Refactor Bench: Can Coding Agents Complete a Long-Horizon, Whole-Repository Stack Migration?",
      "authors": "Deyao Hong, Yizhe Chi, Wenyi Li, Xiaoqiu Wang, Mingju Gao, Kaisen Yang, Bingxiang He, Youjie Zheng, Calvin Xiao, Qinhuai Na",
      "year": "2026",
      "tasks": "20 whole-repository stack migrations",
      "format": "Migration audit, frozen behavioral checks, and agentic verification",
      "difficulty": "6- to 30-hour autonomous repository migrations",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/swe-refactor-bench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/swe-refactor-bench.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "ai4aiBench",
      "name": "AI4AI-Bench",
      "fullName": "AI4AI-Bench: Benchmarking LLM Agents in Algorithmic Design for Recursive Self-Improvement",
      "description": "Tests whether coding agents can improve the training algorithm inside an existing AI research codebase, then survive a sealed training run and held-out evaluation.",
      "paperUrl": "https://arxiv.org/abs/2608.20318",
      "paperTitle": "AI4AI-Bench: Benchmarking LLM Agents in Algorithmic Design for Recursive Self-Improvement",
      "authors": "Yizhe Chi, Wenyi Li, Deyao Hong, Xiaoqiu Wang, Mingju Gao, Kaisen Yang, Bingxiang He, Youjie Zheng, Calvin Xiao, Qinhuai Na",
      "year": "2026",
      "tasks": "10 AI training-algorithm design tasks",
      "format": "Four-hour code rewrite followed by sealed training and held-out evaluation",
      "difficulty": "End-to-end AI research and algorithm design",
      "decimals": 3,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/ai4ai-bench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/ai4ai-bench.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "inferenceBench",
      "name": "InferenceBench",
      "fullName": "InferenceBench",
      "description": "A benchmark for open-ended LLM inference optimization by AI agents. Agents receive a base model, one H100, and a fixed time budget to build a valid OpenAI-compatible inference server that improves serving speed.",
      "paperUrl": "https://inferencebench.ai/",
      "paperTitle": "InferenceBench",
      "authors": "Jehyeok Yeon, Ben Rank, Maksym Andriushchenko",
      "year": "2026",
      "tasks": "4 inference-serving optimization scenarios",
      "format": "Two-hour autonomous CLI agent run",
      "difficulty": "Open-ended ML systems engineering",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/inferencebench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/inferencebench.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "edgeBench",
      "name": "EdgeBench",
      "fullName": "EdgeBench",
      "description": "A ByteDance Seed benchmark of 134 real-world, day-scale tasks that measures how autonomous agents learn from environment feedback over 12+ hour interaction horizons, spanning scientific and ML, systems and software engineering, optimization, knowledge, formal, and game domains.",
      "paperUrl": "https://edge-bench.org/paper.pdf",
      "paperTitle": "EdgeBench technical report",
      "authors": "ByteDance Seed",
      "year": "2026",
      "tasks": "134 tasks (51 public) across 6 domains",
      "format": "Long-horizon interactive agent evaluation",
      "difficulty": "Day-scale expert tasks",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/edgebench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/edgebench.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "bfclV4",
      "name": "BFCL v4",
      "fullName": "Berkeley Function Calling Leaderboard v4",
      "description": "A function-calling benchmark for tool selection, schema adherence, and argument correctness.",
      "paperUrl": "https://www.arcee.ai/blog/trinity-large-thinking",
      "paperTitle": "Trinity-Large-Thinking: Scaling an Open Source Frontier Agent",
      "authors": "Arcee AI",
      "year": "2026",
      "tasks": "Function-calling tasks",
      "format": "Tool invocation and schema evaluation",
      "difficulty": "Advanced tool use",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 14,
      "url": "https://benchlm.ai/benchmarks/bfcl-v4",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/bfcl-v4.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "mleBenchLite",
      "name": "MLE-Bench Lite",
      "fullName": "MLE-Bench Lite",
      "description": "A lightweight machine-learning competition benchmark that measures whether models can iteratively train, evaluate, and improve ML systems in low-resource settings.",
      "paperUrl": "https://www.minimax.io/news/minimax-m27-en",
      "paperTitle": "MiniMax M2.7: Early Echoes of Self-Evolution",
      "authors": "MiniMax",
      "year": "2026",
      "tasks": "Low-resource ML competitions",
      "format": "Autonomous iterative ML optimization",
      "difficulty": "Agentic machine learning",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/mle-bench-lite",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mle-bench-lite.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "mmClawBench",
      "name": "MM-ClawBench",
      "fullName": "MM-ClawBench",
      "description": "An OpenClaw-derived agent benchmark covering practical work and life tasks such as office document delivery, research, planning, and code maintenance.",
      "paperUrl": "https://www.minimax.io/news/minimax-m27-en",
      "paperTitle": "MiniMax M2.7: Early Echoes of Self-Evolution",
      "authors": "MiniMax",
      "year": "2026",
      "tasks": "OpenClaw-style real-world tasks",
      "format": "Agent workflow evaluation",
      "difficulty": "Broad real-world agentic execution",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/mmclawbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mmclawbench.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "clawEval",
      "name": "Claw-Eval",
      "fullName": "Claw-Eval",
      "description": "A transparent real-world autonomous-agent benchmark with 300 human-verified tasks, 2,159 rubric items, and Pass^3 scoring across general, multi-turn, and native multimodal agent tasks.",
      "paperUrl": "https://arxiv.org/abs/2604.06132",
      "paperTitle": "Claw-Eval: Towards Trustworthy Evaluation of Autonomous Agents",
      "authors": "Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang",
      "year": "2026",
      "tasks": "300 tasks, 2,159 rubrics",
      "format": "End-to-end autonomous-agent evaluation with Pass^3 scoring",
      "difficulty": "Real-world general, multi-turn, and native multimodal agent execution",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 36,
      "url": "https://benchlm.ai/benchmarks/claw-eval",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/claw-eval.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "researchClawBench",
      "name": "ResearchClawBench",
      "fullName": "ResearchClawBench",
      "description": "An end-to-end autonomous scientific research benchmark with 40 tasks across 10 scientific domains, where agents receive related literature and raw data, then attempt to rediscover the hidden target paper.",
      "paperUrl": "https://arxiv.org/abs/2606.07591",
      "paperTitle": "ResearchClawBench: A Benchmark for End-to-End Autonomous Scientific Research",
      "authors": "InternScience",
      "year": "2026",
      "tasks": "40 tasks across 10 scientific domains",
      "format": "End-to-end autonomous research evaluation with RADS scoring",
      "difficulty": "Scientific research re-discovery",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 19,
      "url": "https://benchlm.ai/benchmarks/researchclawbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/researchclawbench.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "qwenClawBench",
      "name": "QwenClawBench",
      "fullName": "QwenClawBench",
      "description": "Qwen's internal OpenClaw-style benchmark for measuring broad real-world agent performance across practical productivity and research tasks.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2026",
      "tasks": "Real-world agent workflows",
      "format": "End-to-end agent evaluation",
      "difficulty": "Broad real-world agentic execution",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 10,
      "url": "https://benchlm.ai/benchmarks/qwenclawbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/qwenclawbench.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "qwenWebBench",
      "name": "QwenWebBench",
      "fullName": "QwenWebBench",
      "description": "A Qwen benchmark for artifact and webpage generation quality reported as an Elo-style rating.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2026",
      "tasks": "Web artifacts and interactive deliverables",
      "format": "Elo-style artifact benchmark",
      "difficulty": "Artifact generation",
      "decimals": 0,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 5,
      "url": "https://benchlm.ai/benchmarks/qwenwebbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/qwenwebbench.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "tau3Bench",
      "name": "τ³-bench results",
      "fullName": "τ³-Bench Tool-Agent-User Evaluation",
      "description": "τ³-bench is the current evolution of Sierra's tool-agent-user framework, adding corrected task releases and newer knowledge and voice evaluation modes alongside airline, retail, and telecom.",
      "paperUrl": "https://github.com/sierra-research/tau2-bench",
      "paperTitle": "Official τ³-bench repository and release notes",
      "authors": "Sierra Research",
      "year": "2026",
      "tasks": "Corrected customer-service tasks plus knowledge and voice evaluation modes",
      "format": "Published domain or average success results",
      "difficulty": "Long-horizon, multimodal, and knowledge-aware tool use",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 14,
      "url": "https://benchlm.ai/benchmarks/tau3-bench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/tau3-bench.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "vitaBench",
      "name": "VITA-Bench",
      "fullName": "VITA-Bench",
      "description": "An interactive real-world agent benchmark grounded in practical consumer-service tasks such as delivery, in-store consumption, and online travel workflows.",
      "paperUrl": "https://vitabench.github.io/",
      "paperTitle": "VitaBench: Benchmarking LLM Agents with Versatile Interactive Tasks in Real-world Applications",
      "authors": "Meituan LongCat Team",
      "year": "2025",
      "tasks": "Interactive consumer-service agent tasks",
      "format": "End-to-end interactive agent evaluation",
      "difficulty": "Long-horizon real-world workflows",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 10,
      "url": "https://benchlm.ai/benchmarks/vitabench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/vitabench.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "deepPlanning",
      "name": "DeepPlanning",
      "fullName": "DeepPlanning",
      "description": "A long-horizon planning benchmark that tests whether agents can optimize under explicit time, budget, and feasibility constraints.",
      "paperUrl": "https://arxiv.org/abs/2601.18137",
      "paperTitle": "DeepPlanning: Benchmarking Long-Horizon Agentic Planning with Verifiable Constraints",
      "authors": "DeepPlanning authors",
      "year": "2026",
      "tasks": "Travel planning and constrained shopping",
      "format": "Long-horizon planning benchmark",
      "difficulty": "Constrained agent planning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 7,
      "url": "https://benchlm.ai/benchmarks/deepplanning",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/deepplanning.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "mcpTasks",
      "name": "MCP-Tasks",
      "fullName": "MCP-Tasks",
      "description": "A Model Context Protocol task benchmark used in Qwen's launch tables to measure practical execution over MCP-style tools and integrations.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2026",
      "tasks": "MCP-integrated tool tasks",
      "format": "Interactive tool-use evaluation",
      "difficulty": "Advanced MCP workflows",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 5,
      "url": "https://benchlm.ai/benchmarks/mcptasks",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mcptasks.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "wideResearch",
      "name": "WideResearch",
      "fullName": "WideResearch",
      "description": "A broad research-agent benchmark for open-ended information gathering, synthesis, and answer construction across wide search spaces.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2026",
      "tasks": "Open-ended research tasks",
      "format": "Multi-source research evaluation",
      "difficulty": "Broad research-agent workflows",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 14,
      "url": "https://benchlm.ai/benchmarks/wideresearch",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/wideresearch.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "coworkBench",
      "name": "CoWorkBench",
      "fullName": "CoWorkBench",
      "description": "Qwen's internal benchmark for long-horizon professional work across computer science, finance, law, medicine, and other productivity domains.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.8",
      "paperTitle": "Qwen3.8-Max: A New Bar for Coding and Cowork",
      "authors": "Qwen Team",
      "year": "2026",
      "tasks": "Long-horizon professional workflows",
      "format": "Provider-run agent score",
      "difficulty": "Cross-domain professional work",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 3,
      "url": "https://benchlm.ai/benchmarks/coworkbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/coworkbench.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "mobileWorld",
      "name": "MobileWorld",
      "fullName": "MobileWorld",
      "description": "A mobile-use agent benchmark for completing interactive tasks in smartphone environments.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.8",
      "paperTitle": "Qwen3.8-Max: A New Bar for Coding and Cowork",
      "authors": "Qwen Team",
      "year": "2026",
      "tasks": "Interactive mobile-device workflows",
      "format": "Mobile agent task score",
      "difficulty": "Long-horizon mobile computer use",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/mobileworld",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mobileworld.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "gaia",
      "name": "GAIA",
      "fullName": "General AI Assistants",
      "description": "GAIA evaluates AI models on real-world tasks that are conceptually simple for humans but require multi-step reasoning, web browsing, tool use, and multimodal understanding for AI. Tasks span three difficulty levels and test practical assistant capabilities rather than academic knowledge.",
      "paperUrl": null,
      "paperTitle": null,
      "authors": null,
      "year": 2024,
      "tasks": 466,
      "format": null,
      "difficulty": null,
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/gaia",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/gaia.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "tauBench",
      "name": "TAU-bench",
      "fullName": "Tool-Agent-User Benchmark",
      "description": "Original TAU-bench evaluates a model-driven agent in simulated airline and retail customer-service conversations with domain tools, database state, and policy constraints.",
      "paperUrl": "https://arxiv.org/abs/2406.12045",
      "paperTitle": "τ-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains",
      "authors": "Shunyu Yao, Noah Shinn, Pedram Razavi, Karthik Narasimhan",
      "year": "2024",
      "tasks": "Airline and retail task sets in the archived 2024 release",
      "format": "Domain-specific pass^1 through pass^4 task success",
      "difficulty": "Policy-constrained, multi-turn customer service",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/tau-bench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/tau-bench.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "webArena",
      "name": "WebArena",
      "fullName": "WebArena Web Agent Benchmark",
      "description": "WebArena tests whether a browser-agent system can complete 812 long-horizon tasks inside self-hosted replicas of functional websites. It checks the requested end state, so a result reflects the model, agent scaffold, browser interface, action budget, and evaluator together—not the base model alone.",
      "paperUrl": "https://arxiv.org/abs/2307.13854",
      "paperTitle": "WebArena: A Realistic Web Environment for Building Autonomous Agents",
      "authors": "Shuyan Zhou, Frank F. Xu, Hao Zhu, Xuhui Zhou, Robert Lo, Abishek Sridhar, Xianyi Cheng, Tianyue Ou, Yonatan Bisk, Daniel Fried, Uri Alon, Graham Neubig",
      "year": "2024",
      "tasks": "812 long-horizon browser tasks",
      "format": "End-state task success",
      "difficulty": "Stateful multi-site browser work",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/webarena",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/webarena.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "webArenaVerified",
      "name": "WebArena-Verified",
      "fullName": "WebArena-Verified Browser Agent Benchmark",
      "description": "WebArena-Verified is an audited release of the WebArena browser-agent benchmark. It rechecks task descriptions, reference answers, and evaluators, and replaces nondeterministic judging with deterministic checks where possible.",
      "paperUrl": "https://openreview.net/forum?id=94tlGxmqkN",
      "paperTitle": "WebArena-Verified: A Fully Audited Benchmark for Web Agents",
      "authors": "Amine El Hattami, Megh Thakkar, Nicolas Chapados, Christopher Pal",
      "year": "2025",
      "tasks": "812 verified tasks; separate 258-task Hard subset",
      "format": "Deterministic end-state task success",
      "difficulty": "Audited stateful browser work",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 3,
      "url": "https://benchlm.ai/benchmarks/webarena-verified",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/webarena-verified.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "mewc",
      "name": "MEWC",
      "fullName": "Multi-Environment Web Challenge",
      "description": "A benchmark that evaluates AI agents on multi-environment web challenges, testing navigation and task completion across diverse live web environments.",
      "paperUrl": "https://www.minimax.io/news/minimax-m25",
      "paperTitle": "MiniMax M2.5 benchmark release surface",
      "authors": "MiniMax / benchmark maintainers",
      "year": "2026",
      "tasks": "Web-agent tasks",
      "format": "Browser task completion",
      "difficulty": "Open-web agent workflows",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/mewc",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mewc.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "financeAgentV2",
      "name": "Finance Agent v2",
      "fullName": "Finance Agent v2",
      "description": "Vals AI benchmark for realistic financial analyst agent tasks across qualitative analysis, quantitative analysis, market work, comparables, precedents, earnings, disclosure, and modeling.",
      "paperUrl": "https://www.vals.ai/benchmarks/fabv2",
      "paperTitle": "Finance Agent v2",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "Financial analyst task categories",
      "format": "Mean score across repeated runs",
      "difficulty": "Professional expert-task agent workflow",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 4,
      "url": "https://benchlm.ai/benchmarks/financeagentv2",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/financeagentv2.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "marketBench",
      "name": "Market-Bench",
      "fullName": "Market-Bench",
      "description": "A quantitative-trading implementation benchmark that asks models to build backtesters under market-book liquidity and execution-delay constraints, then compares their outputs with a verifier.",
      "paperUrl": "https://arxiv.org/abs/2512.12264",
      "paperTitle": "Market-Bench",
      "authors": "AfterQuery",
      "year": "2025",
      "tasks": "3 quantitative-trading strategies",
      "format": "Backtester implementation scored by mean absolute error",
      "difficulty": "Market simulation and quantitative coding",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/marketbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/marketbench.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "gdpvalRubrics",
      "name": "GDPval rubrics",
      "fullName": "GDPval rubrics",
      "description": "A display-only provider-table GDPval rubric score for economically valuable work tasks.",
      "paperUrl": "https://huggingface.co/MiniMaxAI/MiniMax-M3",
      "paperTitle": "MiniMax M3 model card",
      "authors": "MiniMax",
      "year": "2026",
      "tasks": "Economically valuable work tasks",
      "format": "Rubric score",
      "difficulty": "Professional agentic workflows",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/gdpvalrubrics",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/gdpvalrubrics.md"
    },
    {
      "category": "agentic",
      "categoryLabel": "Agentic",
      "benchmarkKey": "bankerToolBench",
      "name": "BankerToolBench",
      "fullName": "BankerToolBench",
      "description": "A display-only provider benchmark for finance-oriented tool-use and agent workflows.",
      "paperUrl": "https://huggingface.co/MiniMaxAI/MiniMax-M3",
      "paperTitle": "MiniMax M3 model card",
      "authors": "MiniMax",
      "year": "2026",
      "tasks": "Finance and banking tool-use tasks",
      "format": "Task success rate",
      "difficulty": "Professional finance-agent workflows",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/bankertoolbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/bankertoolbench.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "chartography",
      "name": "Chartography (no tools)",
      "fullName": "Chartography without tools",
      "description": "Professional chart understanding across 100 specialized chart types with expert-set answer tolerances.",
      "paperUrl": "https://surgehq.ai/blog/chartography",
      "paperTitle": "Chartography",
      "authors": "Surge AI and Anthropic",
      "year": "2026",
      "tasks": "100 specialized chart types",
      "format": "Accuracy without tools",
      "difficulty": "Professional chart reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/chartography",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/chartography.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "chartographyWithTools",
      "name": "Chartography (tools)",
      "fullName": "Chartography with image and code tools",
      "description": "Professional chart understanding with a container, standard libraries, and image cropping.",
      "paperUrl": "https://surgehq.ai/blog/chartography",
      "paperTitle": "Chartography",
      "authors": "Surge AI and Anthropic",
      "year": "2026",
      "tasks": "100 specialized chart types",
      "format": "Accuracy with tools",
      "difficulty": "Professional chart reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/chartographywithtools",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/chartographywithtools.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "benchCadVision2Code",
      "name": "BenchCAD Vision2Code (no tools)",
      "fullName": "BenchCAD Vision2Code voxel IoU without tools",
      "description": "Generates CadQuery code from multi-view renders and scores geometric similarity by voxel intersection-over-union.",
      "paperUrl": "https://arxiv.org/abs/2605.10865",
      "paperTitle": "BenchCAD: A comprehensive, industry-standard benchmark for programmatic CAD",
      "authors": "Zhang et al. and Anthropic",
      "year": "2026",
      "tasks": "1,000-file Vision2Code subset",
      "format": "Voxel IoU",
      "difficulty": "Programmatic CAD generation",
      "decimals": 3,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/benchcadvision2code",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/benchcadvision2code.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "benchCadVision2CodeWithTools",
      "name": "BenchCAD Vision2Code (tools)",
      "fullName": "BenchCAD Vision2Code voxel IoU with tools",
      "description": "Generates CadQuery code from multi-view renders with image inspection and code-execution tools.",
      "paperUrl": "https://arxiv.org/abs/2605.10865",
      "paperTitle": "BenchCAD: A comprehensive, industry-standard benchmark for programmatic CAD",
      "authors": "Zhang et al. and Anthropic",
      "year": "2026",
      "tasks": "1,000-file Vision2Code subset",
      "format": "Voxel IoU with tools",
      "difficulty": "Programmatic CAD generation",
      "decimals": 3,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/benchcadvision2codewithtools",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/benchcadvision2codewithtools.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "gdpPdf",
      "name": "GDP.pdf (no tools)",
      "fullName": "GDP.pdf mean criteria pass rate without tools",
      "description": "Professional document understanding over 100 real-world PDFs from ten domains.",
      "paperUrl": "https://surgehq.ai/blog/gdp-pdf-can-100b-ai-models-master-the-documents-that-run-the-world",
      "paperTitle": "GDP.pdf",
      "authors": "Surge AI and Anthropic",
      "year": "2026",
      "tasks": "100 professional document prompts",
      "format": "Mean criteria pass rate",
      "difficulty": "Professional document reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/gdppdf",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/gdppdf.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "gdpPdfWithTools",
      "name": "GDP.pdf (tools)",
      "fullName": "GDP.pdf mean criteria pass rate with tools",
      "description": "Professional document understanding with a container, standard libraries, and image cropping.",
      "paperUrl": "https://surgehq.ai/blog/gdp-pdf-can-100b-ai-models-master-the-documents-that-run-the-world",
      "paperTitle": "GDP.pdf",
      "authors": "Surge AI and Anthropic",
      "year": "2026",
      "tasks": "100 professional document prompts",
      "format": "Mean criteria pass rate with tools",
      "difficulty": "Professional document reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/gdppdfwithtools",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/gdppdfwithtools.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "officeQa",
      "name": "OfficeQA",
      "fullName": "OfficeQA",
      "description": "Grounded numerical reasoning over a corpus of historical U.S. Treasury Bulletin documents.",
      "paperUrl": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf",
      "paperTitle": "Claude Opus 5 System Card",
      "authors": "Databricks and Anthropic",
      "year": "2026",
      "tasks": "Historical Treasury Bulletin questions",
      "format": "Agentic grounded QA accuracy",
      "difficulty": "Professional document reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/officeqa",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/officeqa.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "mmmu",
      "name": "MMMU",
      "fullName": "Massive Multi-discipline Multimodal Understanding",
      "description": "A broad multimodal reasoning benchmark spanning charts, diagrams, tables, and academic visual question answering.",
      "paperUrl": "https://arxiv.org/abs/2401.05508",
      "paperTitle": "MMMU: A Massive Multi-discipline Multimodal Understanding and Reasoning Benchmark for Expert AGI",
      "authors": "MMMU authors",
      "year": "2024",
      "tasks": "Multimodal academic reasoning",
      "format": "Image + text question answering",
      "difficulty": "Frontier multimodal",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 9,
      "url": "https://benchlm.ai/benchmarks/mmmu",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mmmu.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "mmmuPro",
      "name": "MMMU-Pro",
      "fullName": "Massive Multi-discipline Multimodal Understanding Pro",
      "description": "A harder multimodal benchmark for frontier models that combines text with images, diagrams, charts, and academic visual reasoning tasks.",
      "paperUrl": "https://arxiv.org/abs/2409.02813",
      "paperTitle": "MMMU-Pro: A More Robust Multi-discipline Multimodal Understanding Benchmark",
      "authors": "MMMU-Pro authors",
      "year": "2024",
      "tasks": "Multimodal academic reasoning",
      "format": "Image + text question answering",
      "difficulty": "Frontier multimodal",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": 0.45,
      "displayableScoreCount": 38,
      "url": "https://benchlm.ai/benchmarks/mmmu-pro",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mmmu-pro.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "aaMmmuPro",
      "name": "AA-MMMU-Pro",
      "fullName": "Artificial Analysis MMMU-Pro",
      "description": "A display-only Artificial Analysis MMMU-Pro score.",
      "paperUrl": "https://artificialanalysis.ai/evaluations/mmmu-pro",
      "paperTitle": "Artificial Analysis MMMU-Pro Benchmark Leaderboard",
      "authors": "Artificial Analysis",
      "year": "2026",
      "tasks": "Multimodal academic reasoning",
      "format": "Image + text question answering",
      "difficulty": "Frontier multimodal",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 91,
      "url": "https://benchlm.ai/benchmarks/aammmupro",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/aammmupro.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "ocrBenchV2",
      "name": "OCRBench V2",
      "fullName": "OCRBench V2",
      "description": "A native OCR benchmark for reading text from images across multilingual scripts, low-quality scans, handwriting, structured layouts, charts, and screenshots.",
      "paperUrl": "https://arxiv.org/abs/2501.00321",
      "paperTitle": "OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning",
      "authors": "OCRBench authors",
      "year": "2025",
      "tasks": "Image OCR tasks",
      "format": "Accuracy",
      "difficulty": "Native visual text understanding",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 3,
      "url": "https://benchlm.ai/benchmarks/ocrbenchv2",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/ocrbenchv2.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "olmOcr",
      "name": "olmOCR",
      "fullName": "olmOCR-Bench",
      "description": "An end-to-end document understanding benchmark over long, layout-rich PDFs with tables, equations, headers, footnotes, and multi-column flows.",
      "paperUrl": "https://github.com/allenai/olmocr/tree/main/olmocr/bench",
      "paperTitle": "olmOCR-Bench",
      "authors": "Allen Institute for AI",
      "year": "2025",
      "tasks": "Layout-rich PDF understanding",
      "format": "Mean accuracy",
      "difficulty": "Complex document processing",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/olmocr",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/olmocr.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "voxPopuliWer",
      "name": "VoxPopuli WER",
      "fullName": "VoxPopuli-Cleaned-AA Word Error Rate",
      "description": "A speech-recognition benchmark on the cleaned Artificial Analysis VoxPopuli subset, reported as word error rate where lower is better.",
      "paperUrl": "https://huggingface.co/datasets/ArtificialAnalysis/VoxPopuli-Cleaned-AA",
      "paperTitle": "VoxPopuli-Cleaned-AA",
      "authors": "Artificial Analysis / VoxPopuli dataset authors",
      "year": "2026",
      "tasks": "Speech-to-text transcription",
      "format": "Word error rate",
      "difficulty": "Audio speech recognition",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/voxpopuliwer",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/voxpopuliwer.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "designArenaWebsite",
      "name": "Design Arena Website",
      "fullName": "Design Arena Website Elo",
      "description": "A display-only Design Arena website-generation Elo score surfaced on OpenRouter model benchmark pages.",
      "paperUrl": "https://openrouter.ai/x-ai/grok-4.3/benchmarks",
      "paperTitle": "OpenRouter Grok 4.3 benchmarks",
      "authors": "Design Arena",
      "year": "2026",
      "tasks": "Website generation comparisons",
      "format": "Elo",
      "difficulty": "Design and website generation",
      "decimals": 0,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 86,
      "url": "https://benchlm.ai/benchmarks/designarenawebsite",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/designarenawebsite.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "officeQaPro",
      "name": "OfficeQA Pro",
      "fullName": "OfficeQA Pro",
      "description": "A benchmark for grounded reasoning over office-style documents, spreadsheets, charts, and business artifacts.",
      "paperUrl": "https://arxiv.org/abs/2603.08655",
      "paperTitle": "OfficeQA Pro: An Enterprise Benchmark for End-to-End Grounded Reasoning",
      "authors": "OfficeQA Pro authors",
      "year": "2026",
      "tasks": "Document and spreadsheet tasks",
      "format": "Grounded QA over office artifacts",
      "difficulty": "Enterprise grounded reasoning",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": 0.3,
      "displayableScoreCount": 10,
      "url": "https://benchlm.ai/benchmarks/officeqapro",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/officeqapro.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "mathVisionPython",
      "name": "MathVision w/ Python",
      "fullName": "MathVision with Python",
      "description": "A tool-augmented MathVision variant that permits Python during visual mathematics reasoning.",
      "paperUrl": "https://www.kimi.com/blog/kimi-k3",
      "paperTitle": "Kimi K3: Open Frontier Intelligence",
      "authors": "Moonshot AI / MathVision authors",
      "year": "2026",
      "tasks": "Visual mathematics problems with Python",
      "format": "Image and mathematics reasoning with tools",
      "difficulty": "Advanced multimodal mathematics",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 4,
      "url": "https://benchlm.ai/benchmarks/mathvisionpython",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mathvisionpython.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "babyVisionPython",
      "name": "BabyVision w/ Python",
      "fullName": "BabyVision with Python",
      "description": "A Python-assisted BabyVision evaluation for fine-grained visual perception and grounded reasoning.",
      "paperUrl": "https://www.kimi.com/blog/kimi-k3",
      "paperTitle": "Kimi K3: Open Frontier Intelligence",
      "authors": "Moonshot AI",
      "year": "2026",
      "tasks": "Visual perception tasks with Python",
      "format": "Tool-augmented multimodal score",
      "difficulty": "Fine-grained visual perception",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 3,
      "url": "https://benchlm.ai/benchmarks/babyvisionpython",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/babyvisionpython.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "zeroBenchPython",
      "name": "ZeroBench w/ Python",
      "fullName": "ZeroBench_main with Python",
      "description": "A Python-assisted ZeroBench_main evaluation reported as pass@5.",
      "paperUrl": "https://www.kimi.com/blog/kimi-k3",
      "paperTitle": "Kimi K3: Open Frontier Intelligence",
      "authors": "Moonshot AI / ZeroBench authors",
      "year": "2026",
      "tasks": "Visual reasoning questions with Python",
      "format": "Pass@5",
      "difficulty": "Tool-augmented visual reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/zerobenchpython",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/zerobenchpython.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "worldVqaForceAnswer",
      "name": "WorldVQA ForceAnswer",
      "fullName": "WorldVQA ForceAnswer",
      "description": "A forced-answer WorldVQA variant for atomic visual world knowledge.",
      "paperUrl": "https://www.kimi.com/blog/kimi-k3",
      "paperTitle": "Kimi K3: Open Frontier Intelligence",
      "authors": "Moonshot AI / WorldVQA authors",
      "year": "2026",
      "tasks": "Atomic visual world-knowledge questions",
      "format": "Forced-answer visual QA",
      "difficulty": "Fine-grained visual knowledge",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/worldvqaforceanswer",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/worldvqaforceanswer.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "omniDocBench",
      "name": "OmniDocBench",
      "fullName": "OmniDocBench",
      "description": "A document-understanding benchmark for parsing and reasoning over complex document layouts.",
      "paperUrl": "https://www.kimi.com/blog/kimi-k3",
      "paperTitle": "Kimi K3: Open Frontier Intelligence",
      "authors": "Moonshot AI / OmniDocBench authors",
      "year": "2026",
      "tasks": "Complex document-understanding tasks",
      "format": "Document-understanding score",
      "difficulty": "Grounded document reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/omnidocbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/omnidocbench.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "perceptionBench",
      "name": "PerceptionBench",
      "fullName": "PerceptionBench (Internal)",
      "description": "Moonshot AI's internal benchmark for atomic visual perception capabilities.",
      "paperUrl": "https://www.kimi.com/blog/kimi-k3",
      "paperTitle": "Kimi K3: Open Frontier Intelligence",
      "authors": "Moonshot AI",
      "year": "2026",
      "tasks": "Internal atomic visual-perception tasks",
      "format": "Internal evaluation score",
      "difficulty": "Fine-grained visual perception",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 3,
      "url": "https://benchlm.ai/benchmarks/perceptionbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/perceptionbench.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "mmmuProPython",
      "name": "MMMU-Pro w/ Python",
      "fullName": "MMMU-Pro with Python",
      "description": "Tool-augmented MMMU-Pro variant that allows Python assistance during multimodal reasoning.",
      "paperUrl": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
      "paperTitle": "Introducing GPT-5.4 mini and nano",
      "authors": "OpenAI",
      "year": "2026",
      "tasks": "Multimodal academic reasoning",
      "format": "Image + text question answering with Python",
      "difficulty": "Frontier multimodal",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 9,
      "url": "https://benchlm.ai/benchmarks/mmmupropython",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mmmupropython.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "omniDocBench15",
      "name": "OmniDocBench 1.5",
      "fullName": "OmniDocBench 1.5",
      "description": "A document understanding benchmark used in frontier-model comparison tables to measure extraction and grounded reasoning quality on complex documents.",
      "paperUrl": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
      "paperTitle": "Introducing GPT-5.4 mini and nano",
      "authors": "OpenAI",
      "year": "2026",
      "tasks": "Document understanding tasks",
      "format": "Document understanding benchmark",
      "difficulty": "Grounded document reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 6,
      "url": "https://benchlm.ai/benchmarks/omnidocbench15",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/omnidocbench15.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "liquidExtractJsonValidity",
      "name": "Liquid Extract JSON Validity",
      "fullName": "Liquid image-to-JSON extraction JSON validity",
      "description": "A display-only Liquid AI extraction metric measuring the share of image-to-JSON outputs that parse as strict JSON.",
      "paperUrl": "https://huggingface.co/LiquidAI/LFM2.5-VL-1.6B-Extract",
      "paperTitle": "LiquidAI LFM2.5-VL Extract model cards",
      "authors": "Liquid AI",
      "year": "2026",
      "tasks": "Image-to-JSON extraction",
      "format": "Strict JSON parseability rate",
      "difficulty": "Structured visual extraction",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/liquidextractjsonvalidity",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/liquidextractjsonvalidity.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "liquidExtractSchemaF1",
      "name": "Liquid Extract F1",
      "fullName": "Liquid image-to-JSON extraction schema consistency F1",
      "description": "A display-only Liquid AI extraction metric measuring field-name agreement between requested schema fields and extracted JSON fields.",
      "paperUrl": "https://huggingface.co/LiquidAI/LFM2.5-VL-1.6B-Extract",
      "paperTitle": "LiquidAI LFM2.5-VL Extract model cards",
      "authors": "Liquid AI",
      "year": "2026",
      "tasks": "Image-to-JSON extraction",
      "format": "Schema field F1",
      "difficulty": "Structured visual extraction",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/liquidextractschemaf1",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/liquidextractschemaf1.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "liquidExtractVlmJudge",
      "name": "Liquid Extract VLM Judge",
      "fullName": "Liquid image-to-JSON extraction VLM judge score",
      "description": "A display-only Liquid AI extraction metric measuring judged agreement between extracted values and the source image.",
      "paperUrl": "https://huggingface.co/LiquidAI/LFM2.5-VL-1.6B-Extract",
      "paperTitle": "LiquidAI LFM2.5-VL Extract model cards",
      "authors": "Liquid AI",
      "year": "2026",
      "tasks": "Image-to-JSON extraction",
      "format": "VLM-judged extraction accuracy",
      "difficulty": "Structured visual extraction",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/liquidextractvlmjudge",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/liquidextractvlmjudge.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "realWorldQa",
      "name": "RealWorldQA",
      "fullName": "RealWorldQA",
      "description": "A grounded visual QA benchmark focused on answering practical questions about real-world images and scenes.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2026",
      "tasks": "Real-world visual question answering",
      "format": "Image-grounded QA",
      "difficulty": "General visual reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 7,
      "url": "https://benchlm.ai/benchmarks/realworldqa",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/realworldqa.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "videoMmeWithSub",
      "name": "Video-MME (with subtitle)",
      "fullName": "Video-MME with subtitle",
      "description": "A video understanding benchmark that allows subtitle access when answering multimodal questions about videos.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2026",
      "tasks": "Video understanding",
      "format": "Video QA with subtitle context",
      "difficulty": "Multimodal video reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 6,
      "url": "https://benchlm.ai/benchmarks/videommewithsub",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/videommewithsub.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "videoMmeNoSub",
      "name": "Video-MME (w/o subtitle)",
      "fullName": "Video-MME without subtitle",
      "description": "A stricter Video-MME setting that removes subtitle help and tests video understanding from visual and audio context alone.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2026",
      "tasks": "Video understanding",
      "format": "Video QA without subtitle context",
      "difficulty": "Multimodal video reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/videommenosub",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/videommenosub.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "videoMme",
      "name": "Video-MME",
      "fullName": "Video-MME",
      "description": "A comprehensive benchmark for multimodal large language models on video understanding, covering temporal reasoning, perception, and question answering over videos.",
      "paperUrl": "https://mme-benchmark.github.io/",
      "paperTitle": "Video-MME benchmark",
      "authors": "Video-MME benchmark team",
      "year": "2024",
      "tasks": "Video understanding",
      "format": "Video QA and analysis",
      "difficulty": "Broad multimodal video reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/videomme",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/videomme.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "mathVision",
      "name": "MathVision",
      "fullName": "MathVision",
      "description": "A visual mathematics benchmark that tests whether a model can solve math problems grounded in diagrams, equations, figures, and other visual inputs.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2026",
      "tasks": "Visually grounded math problems",
      "format": "Image + math reasoning",
      "difficulty": "Advanced multimodal mathematics",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 16,
      "url": "https://benchlm.ai/benchmarks/mathvision",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mathvision.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "weMath",
      "name": "We-Math",
      "fullName": "We-Math",
      "description": "A multimodal math benchmark for visually grounded mathematical reasoning and answer generation.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2026",
      "tasks": "Visually grounded math problems",
      "format": "Multimodal mathematical reasoning",
      "difficulty": "Advanced multimodal mathematics",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/wemath",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/wemath.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "dynaMath",
      "name": "DynaMath",
      "fullName": "DynaMath",
      "description": "A multimodal benchmark for dynamic mathematical reasoning over visual and structured inputs.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2026",
      "tasks": "Dynamic visual math problems",
      "format": "Multimodal mathematical reasoning",
      "difficulty": "Advanced multimodal mathematics",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/dynamath",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/dynamath.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "mStar",
      "name": "MStar",
      "fullName": "MStar",
      "description": "A general visual question-answering benchmark used in provider tables for real-image reasoning quality.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2026",
      "tasks": "Real-image visual QA",
      "format": "Image-grounded QA",
      "difficulty": "General visual reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/mstar",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mstar.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "chatCvqa",
      "name": "ChatCVQA",
      "fullName": "ChatCVQA",
      "description": "A conversational visual QA benchmark that tests multi-turn grounded answering over images and documents.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2026",
      "tasks": "Conversational visual QA",
      "format": "Multi-turn image-grounded QA",
      "difficulty": "Conversational multimodal reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/chatcvqa",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/chatcvqa.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "mmLongBenchDoc",
      "name": "MMLongBench-Doc",
      "fullName": "MMLongBench-Doc",
      "description": "A long-document multimodal benchmark for grounded reasoning over extended document contexts.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2026",
      "tasks": "Long document understanding",
      "format": "Document-grounded reasoning",
      "difficulty": "Long-context document reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/mmlongbenchdoc",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mmlongbenchdoc.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "ccOcr",
      "name": "CC-OCR",
      "fullName": "CC-OCR",
      "description": "An OCR-focused benchmark for reading and extracting text from visually complex documents and images.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2026",
      "tasks": "Optical character recognition",
      "format": "Text extraction from images and documents",
      "difficulty": "Document reading",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 3,
      "url": "https://benchlm.ai/benchmarks/ccocr",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/ccocr.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "ai2dTest",
      "name": "AI2D_TEST",
      "fullName": "AI2D test split",
      "description": "A diagram understanding benchmark focused on scientific and educational visual question answering.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2026",
      "tasks": "Diagram understanding",
      "format": "Diagram-grounded QA",
      "difficulty": "Structured visual reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/ai2dtest",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/ai2dtest.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "countBench",
      "name": "CountBench",
      "fullName": "CountBench",
      "description": "A visual counting benchmark that tests whether a model can count objects and entities reliably in complex scenes.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2026",
      "tasks": "Visual counting tasks",
      "format": "Image-grounded counting",
      "difficulty": "Fine-grained visual perception",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/countbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/countbench.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "refcocoAvg",
      "name": "RefCOCO (avg)",
      "fullName": "RefCOCO average",
      "description": "A referring-expression grounding benchmark averaged across RefCOCO variants to test whether a model can localize described objects correctly.",
      "paperUrl": "https://github.com/lichengunc/refer",
      "paperTitle": "RefCOCO referring expression datasets",
      "authors": "RefCOCO dataset authors",
      "year": "2026",
      "tasks": "Referring-expression grounding",
      "format": "Grounded visual localization",
      "difficulty": "Fine-grained visual grounding",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 4,
      "url": "https://benchlm.ai/benchmarks/refcocoavg",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/refcocoavg.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "odinw13",
      "name": "ODINW13",
      "fullName": "ODINW13",
      "description": "A visual detection and grounding benchmark slice used to compare zero-shot object understanding across diverse domains.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2026",
      "tasks": "Out-of-distribution object understanding",
      "format": "Detection and grounding",
      "difficulty": "Robust visual grounding",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/odinw13",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/odinw13.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "erqa",
      "name": "ERQA",
      "fullName": "ERQA",
      "description": "A grounded visual reasoning benchmark focused on evidence-based question answering over real images.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2026",
      "tasks": "Evidence-based visual QA",
      "format": "Grounded image reasoning",
      "difficulty": "Grounded multimodal reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 10,
      "url": "https://benchlm.ai/benchmarks/erqa",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/erqa.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "videoMmmu",
      "name": "VideoMMMU",
      "fullName": "VideoMMMU",
      "description": "A video extension of MMMU-style multimodal reasoning over expert questions grounded in temporal media.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2026",
      "tasks": "Video-grounded expert reasoning",
      "format": "Video + text reasoning",
      "difficulty": "Frontier multimodal video reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 11,
      "url": "https://benchlm.ai/benchmarks/videommmu",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/videommmu.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "mlvuAvg",
      "name": "MLVU (M-Avg)",
      "fullName": "MLVU mean average",
      "description": "A multi-task video understanding benchmark averaged across MLVU categories.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2026",
      "tasks": "General video understanding",
      "format": "Video QA and understanding",
      "difficulty": "Broad multimodal video reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 4,
      "url": "https://benchlm.ai/benchmarks/mlvuavg",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mlvuavg.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "lvBench",
      "name": "LVBench",
      "fullName": "LVBench",
      "description": "A long-video understanding benchmark for retrieving and reasoning over information distributed across extended video inputs.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.8",
      "paperTitle": "Qwen3.8-Max: A New Bar for Coding and Cowork",
      "authors": "Qwen Team",
      "year": "2026",
      "tasks": "Long-form video question answering",
      "format": "Long-video understanding score",
      "difficulty": "Extended temporal reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/lvbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/lvbench.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "mmvu",
      "name": "MMVU",
      "fullName": "Multimodal Multi-disciplinary Video Understanding",
      "description": "A benchmark for evaluating multimodal models on video understanding tasks across multiple disciplines, emphasizing temporal reasoning and comprehension over video content.",
      "paperUrl": "https://www.kimi.com/blog/kimi-k2-5.html",
      "paperTitle": "Kimi K2.5 benchmark release surface",
      "authors": "MMVU benchmark maintainers",
      "year": "2026",
      "tasks": "Video understanding",
      "format": "Video reasoning benchmark",
      "difficulty": "Multi-disciplinary multimodal video reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 7,
      "url": "https://benchlm.ai/benchmarks/mmvu",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mmvu.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "screenSpotPro",
      "name": "ScreenSpot Pro",
      "fullName": "ScreenSpot Pro",
      "description": "A GUI-grounding benchmark for 1,581 instructions in full-screen, high-resolution professional interfaces. It tests where a target is, not whether an agent can finish the surrounding workflow.",
      "paperUrl": "https://arxiv.org/abs/2504.07981",
      "paperTitle": "ScreenSpot-Pro: GUI Grounding for Professional High-Resolution Computer Use",
      "authors": "Kaixin Li, Ziyang Meng, Hongzhan Lin, Ziyang Luo, Yuchen Tian, Jing Ma, Zhiyong Huang, Tat-Seng Chua",
      "year": "2025",
      "tasks": "1,581 grounding instructions",
      "format": "Static interface element localization",
      "difficulty": "Professional GUI grounding",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 17,
      "url": "https://benchlm.ai/benchmarks/screenspot-pro",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/screenspot-pro.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "tirBench",
      "name": "TIR-Bench",
      "fullName": "TIR-Bench",
      "description": "A visual agent benchmark for interface reasoning and task execution over screenshots or software surfaces.",
      "paperUrl": "https://qwen.ai/blog?id=qwen3.6",
      "paperTitle": "Qwen3.6 launch benchmarks",
      "authors": "Qwen",
      "year": "2026",
      "tasks": "Visual agent and interface reasoning",
      "format": "Screenshot-grounded task reasoning",
      "difficulty": "Computer-use visual reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/tirbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/tirbench.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "gdpvalAa",
      "name": "GDPval-AA",
      "fullName": "GDPval-AA",
      "description": "An evaluation focused on professional domain expertise and task delivery quality in office-style knowledge work.",
      "paperUrl": "https://www.minimax.io/news/minimax-m27-en",
      "paperTitle": "MiniMax M2.7: Early Echoes of Self-Evolution",
      "authors": "MiniMax",
      "year": "2026",
      "tasks": "Professional office delivery",
      "format": "ELO-style office benchmark",
      "difficulty": "Professional knowledge work",
      "decimals": 0,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/gdpvalaa",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/gdpvalaa.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "medXpertQaMm",
      "name": "MedXpertQA (MM)",
      "fullName": "MedXpertQA Multimodal",
      "description": "A multimodal medical multiple-choice benchmark covering clinical images such as X-rays, histology, and dermatology.",
      "paperUrl": "https://ai.meta.com/static-resource/muse-spark-eval-methodology",
      "paperTitle": "Muse Spark Eval Methodology",
      "authors": "Meta AI",
      "year": "2026",
      "tasks": "2,000 multimodal medical questions",
      "format": "Medical visual MCQ",
      "difficulty": "Clinical multimodal reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 8,
      "url": "https://benchlm.ai/benchmarks/medxpertqamm",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/medxpertqamm.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "zeroBench",
      "name": "ZeroBench",
      "fullName": "ZeroBench",
      "description": "A multi-step visual reasoning benchmark with pass@5 reporting and optional tool use.",
      "paperUrl": "https://ai.meta.com/static-resource/muse-spark-eval-methodology",
      "paperTitle": "Muse Spark Eval Methodology",
      "authors": "Meta AI",
      "year": "2026",
      "tasks": "100 visual reasoning questions",
      "format": "Multi-step visual reasoning",
      "difficulty": "Tool-augmented visual reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 6,
      "url": "https://benchlm.ai/benchmarks/zerobench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/zerobench.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "design2Code",
      "name": "Design2Code",
      "fullName": "Design2Code",
      "description": "A multimodal coding benchmark for turning visual designs into working frontend implementations.",
      "paperUrl": "https://docs.z.ai/guides/vlm/glm-5v-turbo",
      "paperTitle": "GLM-5V-Turbo",
      "authors": "Z.AI",
      "year": "2026",
      "tasks": "Design-to-code tasks",
      "format": "Visual input to frontend implementation",
      "difficulty": "Multimodal coding",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/design2code",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/design2code.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "flameVlmCode",
      "name": "Flame-VLM-Code",
      "fullName": "Flame-VLM-Code",
      "description": "A vision-language coding benchmark for generating correct code from visual and multimodal inputs.",
      "paperUrl": "https://docs.z.ai/guides/vlm/glm-5v-turbo",
      "paperTitle": "GLM-5V-Turbo",
      "authors": "Z.AI",
      "year": "2026",
      "tasks": "Multimodal coding tasks",
      "format": "Vision-language code generation",
      "difficulty": "Multimodal coding",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/flamevlmcode",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/flamevlmcode.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "vision2Web",
      "name": "Vision2Web",
      "fullName": "Vision2Web",
      "description": "A benchmark for converting visual references into functional web implementations.",
      "paperUrl": "https://docs.z.ai/guides/vlm/glm-5v-turbo",
      "paperTitle": "GLM-5V-Turbo",
      "authors": "Z.AI",
      "year": "2026",
      "tasks": "Screenshot-to-web tasks",
      "format": "Visual reference to web implementation",
      "difficulty": "Multimodal web generation",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 3,
      "url": "https://benchlm.ai/benchmarks/vision2web",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/vision2web.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "imageMining",
      "name": "ImageMining",
      "fullName": "ImageMining",
      "description": "A multimodal retrieval and extraction benchmark over image-heavy task settings.",
      "paperUrl": "https://docs.z.ai/guides/vlm/glm-5v-turbo",
      "paperTitle": "GLM-5V-Turbo",
      "authors": "Z.AI",
      "year": "2026",
      "tasks": "Visual retrieval tasks",
      "format": "Image-grounded retrieval and extraction",
      "difficulty": "Multimodal retrieval",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/imagemining",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/imagemining.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "mmSearch",
      "name": "MMSearch",
      "fullName": "MMSearch",
      "description": "A multimodal search benchmark for retrieval and grounded answering across mixed-media inputs.",
      "paperUrl": "https://docs.z.ai/guides/vlm/glm-5v-turbo",
      "paperTitle": "GLM-5V-Turbo",
      "authors": "Z.AI",
      "year": "2026",
      "tasks": "Multimodal search tasks",
      "format": "Mixed-media retrieval and grounded answering",
      "difficulty": "Multimodal search",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/mmsearch",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mmsearch.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "mmSearchPlus",
      "name": "MMSearch-Plus",
      "fullName": "MMSearch-Plus",
      "description": "A harder MMSearch variant for multimodal retrieval and grounded tool-use workflows.",
      "paperUrl": "https://docs.z.ai/guides/vlm/glm-5v-turbo",
      "paperTitle": "GLM-5V-Turbo",
      "authors": "Z.AI",
      "year": "2026",
      "tasks": "Hard multimodal search tasks",
      "format": "Advanced mixed-media retrieval benchmark",
      "difficulty": "Advanced multimodal search",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/mmsearchplus",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/mmsearchplus.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "simpleVqa",
      "name": "SimpleVQA",
      "fullName": "SimpleVQA",
      "description": "A visual question answering benchmark focused on straightforward image-grounded understanding.",
      "paperUrl": "https://docs.z.ai/guides/vlm/glm-5v-turbo",
      "paperTitle": "GLM-5V-Turbo",
      "authors": "Z.AI",
      "year": "2026",
      "tasks": "Visual QA tasks",
      "format": "Image-grounded question answering",
      "difficulty": "General visual understanding",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 10,
      "url": "https://benchlm.ai/benchmarks/simplevqa",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/simplevqa.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "factsVlm",
      "name": "Facts-VLM",
      "fullName": "Facts-VLM",
      "description": "A grounded multimodal factuality benchmark for evidence-linked answer correctness.",
      "paperUrl": "https://docs.z.ai/guides/vlm/glm-5v-turbo",
      "paperTitle": "GLM-5V-Turbo",
      "authors": "Z.AI",
      "year": "2026",
      "tasks": "Grounded factuality tasks",
      "format": "Evidence-linked multimodal factuality",
      "difficulty": "Grounded multimodal factuality",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/factsvlm",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/factsvlm.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "vStar",
      "name": "V*",
      "fullName": "V*",
      "description": "A vision-centric benchmark for high-level multimodal reasoning and perception quality.",
      "paperUrl": "https://docs.z.ai/guides/vlm/glm-5v-turbo",
      "paperTitle": "GLM-5V-Turbo",
      "authors": "Z.AI",
      "year": "2026",
      "tasks": "Frontier multimodal reasoning tasks",
      "format": "Vision-centric reasoning benchmark",
      "difficulty": "Frontier multimodal",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 11,
      "url": "https://benchlm.ai/benchmarks/vstar",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/vstar.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "charxiv",
      "name": "CharXiv",
      "fullName": "CharXiv Reasoning",
      "description": "A scientific chart reasoning benchmark that tests whether models can understand, interpret, and reason about complex scientific visualizations including plots, diagrams, and data charts.",
      "paperUrl": "https://charxiv.github.io/",
      "paperTitle": "CharXiv: Charting Gaps in Realistic Chart Understanding in Multimodal LLMs",
      "authors": "CharXiv authors",
      "year": "2024",
      "tasks": "Scientific chart reasoning",
      "format": "Chart understanding and reasoning",
      "difficulty": "Scientific visualization reasoning",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": 0.25,
      "displayableScoreCount": 35,
      "url": "https://benchlm.ai/benchmarks/charxiv",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/charxiv.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "charxivNoTools",
      "name": "CharXiv w/o tools",
      "fullName": "CharXiv Reasoning without tools",
      "description": "Tool-free variant of CharXiv that isolates raw visual reasoning ability without code execution or tool augmentation.",
      "paperUrl": "https://charxiv.github.io/",
      "paperTitle": "CharXiv: Charting Gaps in Realistic Chart Understanding in Multimodal LLMs",
      "authors": "CharXiv authors",
      "year": "2024",
      "tasks": "Scientific chart reasoning (tool-free)",
      "format": "Chart understanding without tools",
      "difficulty": "Scientific visualization reasoning",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 11,
      "url": "https://benchlm.ai/benchmarks/charxivnotools",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/charxivnotools.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "babyVision",
      "name": "BabyVision",
      "fullName": "BabyVision",
      "description": "A multimodal benchmark for fine-grained visual perception and grounded reasoning tasks.",
      "paperUrl": "https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report",
      "paperTitle": "Muse Spark 1.1 Evaluation Report",
      "authors": "Meta AI",
      "year": "2026",
      "tasks": "Visual perception tasks",
      "format": "Multimodal visual reasoning",
      "difficulty": "Fine-grained visual perception",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 5,
      "url": "https://benchlm.ai/benchmarks/babyvision",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/babyvision.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "sweMultimodal",
      "name": "SWE-bench Multimodal",
      "fullName": "SWE-bench Multimodal",
      "description": "A multimodal variant of SWE-bench that adds visual context (screenshots, design mockups) to software engineering issue descriptions, testing whether models can leverage visual information for code generation.",
      "paperUrl": "https://www.swebench.com/multimodal",
      "paperTitle": "SWE-bench Multimodal",
      "authors": "SWE-bench team",
      "year": "2025",
      "tasks": "Multimodal software engineering tasks",
      "format": "Code patch generation with visual context",
      "difficulty": "Frontier multimodal coding",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/swe-bench-multimodal",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/swe-bench-multimodal.md"
    },
    {
      "category": "multimodalGrounded",
      "categoryLabel": "Multimodal & Grounded",
      "benchmarkKey": "blueprintBench2",
      "name": "Blueprint-Bench 2",
      "fullName": "Blueprint-Bench 2",
      "description": "An agentic spatial reasoning benchmark reported as a normalized score.",
      "paperUrl": "https://x.com/GoogleDeepMind",
      "paperTitle": "Gemini 3.5 Flash launch screenshots",
      "authors": "Google DeepMind",
      "year": "2026",
      "tasks": "Spatial reasoning from blueprints",
      "format": "Normalized score",
      "difficulty": "Agentic spatial reasoning",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/blueprintbench2",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/blueprintbench2.md"
    },
    {
      "category": "korean",
      "categoryLabel": "korean",
      "benchmarkKey": "kmmlu",
      "name": "KMMLU",
      "fullName": "Korean Massive Multitask Language Understanding",
      "description": "Evaluates Korean expert-level knowledge across 45 subjects. 20% of questions require Korean cultural context.",
      "paperUrl": "https://arxiv.org/abs/2402.11548",
      "paperTitle": "KMMLU: Measuring Massive Multitask Language Understanding in Korean",
      "authors": "KMMLU Authors",
      "year": "2024",
      "tasks": "35,030 questions",
      "format": "Multiple choice questions",
      "difficulty": "Elementary to professional level in Korean",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/kmmlu",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/kmmlu.md"
    },
    {
      "category": "korean",
      "categoryLabel": "korean",
      "benchmarkKey": "kmmluHard",
      "name": "KMMLU-Hard",
      "fullName": "KMMLU-Hard",
      "description": "A filtered hard subset of KMMLU containing ~5,000 questions that most models get wrong.",
      "paperUrl": "https://github.com/daekeun-ml/evaluate-llm-on-korean-dataset",
      "paperTitle": "Evaluating LLMs on Hard Korean Queries",
      "authors": "Daekeun ML",
      "year": "2025",
      "tasks": "~5,000 questions",
      "format": "Multiple choice questions",
      "difficulty": "Advanced Korean reasoning",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/kmmluhard",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/kmmluhard.md"
    },
    {
      "category": "korean",
      "categoryLabel": "korean",
      "benchmarkKey": "kmmluRedux",
      "name": "KMMLU-Redux",
      "fullName": "KMMLU-Redux",
      "description": "Cleaned KMMLU from national technical qualification exams, with errors removed, decontaminated, and deduplicated.",
      "paperUrl": null,
      "paperTitle": null,
      "authors": null,
      "year": null,
      "tasks": "~3,500 questions",
      "format": "Technical multiple choice",
      "difficulty": "Industrial/technical",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/kmmluredux",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/kmmluredux.md"
    },
    {
      "category": "korean",
      "categoryLabel": "korean",
      "benchmarkKey": "kmmluPro",
      "name": "KMMLU-Pro",
      "fullName": "KMMLU-Pro",
      "description": "Korean National Professional Licensure exams evaluating professional-grade knowledge.",
      "paperUrl": null,
      "paperTitle": null,
      "authors": null,
      "year": null,
      "tasks": "~2,500 questions",
      "format": "Professional licensure exams",
      "difficulty": "Professional",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/kmmlupro",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/kmmlupro.md"
    },
    {
      "category": "korean",
      "categoryLabel": "korean",
      "benchmarkKey": "click",
      "name": "CLIcK",
      "fullName": "Cultural and Linguistic Intelligence in Korean",
      "description": "Evaluates Korean culture and linguistics.",
      "paperUrl": null,
      "paperTitle": null,
      "authors": null,
      "year": null,
      "tasks": "1,995 questions",
      "format": "Cultural/linguistic QA",
      "difficulty": "Korean cultural nuances",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/click",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/click.md"
    },
    {
      "category": "korean",
      "categoryLabel": "korean",
      "benchmarkKey": "kobalt",
      "name": "KoBALT",
      "fullName": "Korean Benchmark for Advanced Linguistic Tasks",
      "description": "Evaluates advanced Korean linguistic competence.",
      "paperUrl": null,
      "paperTitle": null,
      "authors": null,
      "year": null,
      "tasks": "Linguistics questions",
      "format": "Advanced linguistics",
      "difficulty": "Advanced linguistic phenomena",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/kobalt",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/kobalt.md"
    },
    {
      "category": "korean",
      "categoryLabel": "korean",
      "benchmarkKey": "koreanCsat",
      "name": "Korean CSAT",
      "fullName": "College Scholastic Ability Test (수능)",
      "description": "The Korean SAT exam.",
      "paperUrl": null,
      "paperTitle": null,
      "authors": null,
      "year": null,
      "tasks": "Multi-subject exam",
      "format": "Standardized test",
      "difficulty": "High school to college level",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/koreancsat",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/koreancsat.md"
    },
    {
      "category": "korean",
      "categoryLabel": "korean",
      "benchmarkKey": "hrm8k",
      "name": "HRM8K",
      "fullName": "HAE-RAE Math 8K",
      "description": "Korean mathematical reasoning (high-school to Olympiad level).",
      "paperUrl": null,
      "paperTitle": null,
      "authors": null,
      "year": null,
      "tasks": "8,011 instances",
      "format": "Math word problems",
      "difficulty": "Olympiad level",
      "decimals": null,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/hrm8k",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/hrm8k.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "kindBench",
      "name": "KindBench",
      "fullName": "KindBench Psychological Safety Benchmark",
      "description": "A behavioral benchmark that tests psychological safety across sixteen adversarial multi-turn conversations covering emotional safety, identity, sycophancy, and value integrity.",
      "paperUrl": "https://www.kindbench.com/methodology",
      "paperTitle": "KindBench methodology",
      "authors": "Aphelion Labs",
      "year": "2026",
      "tasks": "16 multi-turn scenarios, 72 criteria",
      "format": "Judge-scored behavioral audit with human review",
      "difficulty": "Adversarial psychological-safety evaluation",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/kindbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/kindbench.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "liveBench",
      "name": "LiveBench",
      "fullName": "LiveBench",
      "description": "A frequently refreshed benchmark with objective scoring across reasoning, coding, agentic coding, mathematics, data analysis, language, and instruction following.",
      "paperUrl": "https://livebench.ai/",
      "paperTitle": "LiveBench: A Challenging, Contamination-Free LLM Benchmark",
      "authors": "LiveBench",
      "year": "2024",
      "tasks": "23 objective tasks across 7 categories",
      "format": "Mean of category averages",
      "difficulty": "Broad frontier-model evaluation",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/livebench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/livebench.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "valsIndex",
      "name": "Vals Index",
      "fullName": "Vals Index v2",
      "description": "Vals AI composite benchmark across professional finance, coding, modeling, and legal-work tasks, including Finance Agent v2, EMB, Terminal-Bench 2.1, Vibe Code Bench, Code Migration, Legal Research, and HLAB.",
      "paperUrl": "https://www.vals.ai/benchmarks/vals_index",
      "paperTitle": "Vals Index",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "Finance, coding, spreadsheet modeling, code migration, and legal-work components",
      "format": "Composite score",
      "difficulty": "Private economic-work benchmark composite",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/valsindex",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/valsindex.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "valsWebSearchIndex",
      "name": "Web Search Index",
      "fullName": "Vals Web Search Index",
      "description": "A Vals AI comparison of native provider search and Exa across finance-analysis and legal-research tasks.",
      "paperUrl": "https://www.vals.ai/benchmarks/web_search",
      "paperTitle": "Web Search Index",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "Finance Agent Benchmark v2 and Legal Research Benchmark tasks",
      "format": "Accuracy by model and search-tool combination",
      "difficulty": "Professional web research with controlled search-tool variants",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/valswebsearchindex",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/valswebsearchindex.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "valsTimeHorizonKsp",
      "name": "Time Horizon Index: KSP",
      "fullName": "Vals Time Horizon Index: Kerbal Space Program",
      "description": "A Vals AI agent benchmark that gives each system five days to build and run a space program in Kerbal Space Program.",
      "paperUrl": "https://www.vals.ai/benchmarks/time_horizon_index",
      "paperTitle": "Time Horizon Index: KSP",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "30 progressively harder Kerbal Space Program missions",
      "format": "Mission-ladder progress with partial credit",
      "difficulty": "Long-horizon autonomous computer use and planning",
      "decimals": 3,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/valstimehorizonksp",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/valstimehorizonksp.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "valsMultimodalIndex",
      "name": "Vals Multimodal Index",
      "fullName": "Vals Multimodal Index v1.2",
      "description": "Vals AI multimodal composite across finance, coding, education, and mortgage-tax task families.",
      "paperUrl": "https://www.vals.ai/benchmarks/vals_multimodal_index",
      "paperTitle": "Vals Multimodal Index",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "Finance, coding, education, and mortgage-tax components",
      "format": "Composite score",
      "difficulty": "Private multimodal economic-work benchmark composite",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/valsmultimodalindex",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/valsmultimodalindex.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "valsFinanceAgentV1",
      "name": "Finance Agent v1.1",
      "fullName": "Vals Finance Agent v1.1",
      "description": "An archived Vals benchmark covering retrieval, numerical reasoning, financial modeling, market analysis, earnings, trends, and adjustments.",
      "paperUrl": "https://www.vals.ai/benchmarks/finance_agent",
      "paperTitle": "Finance Agent v1.1",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "11 financial analyst task views",
      "format": "Accuracy with task-level breakdowns",
      "difficulty": "Professional financial analysis",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/valsfinanceagentv1",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/valsfinanceagentv1.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "valsReverseEngBench",
      "name": "ReverseEngBench",
      "fullName": "Vals ReverseEngBench",
      "description": "A contamination-resistant agent benchmark for reverse engineering real-world binaries.",
      "paperUrl": "https://www.vals.ai/benchmarks/reverse_eng",
      "paperTitle": "ReverseEngBench",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "Real-world binary reverse-engineering tasks",
      "format": "Fully solved rate and capability score",
      "difficulty": "Agentic reverse engineering",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/valsreverseengbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/valsreverseengbench.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "valsTerminalBench1",
      "name": "Vals Terminal-Bench 1.0 mirror",
      "fullName": "Vals-hosted Terminal-Bench 1.0 mirror",
      "description": "A Vals-hosted view of Terminal-Bench 1.0 with easy, medium, and hard task splits.",
      "paperUrl": "https://www.vals.ai/benchmarks/terminal-bench",
      "paperTitle": "Terminal-Bench 1.0",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "Terminal tasks split by easy, medium, and hard difficulty",
      "format": "Accuracy score",
      "difficulty": "Terminal-agent execution",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/valsterminalbench1",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/valsterminalbench1.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "valsCorpFinV2",
      "name": "CorpFin v2",
      "fullName": "Vals CorpFin v2",
      "description": "Vals AI private benchmark for understanding long-context credit agreements.",
      "paperUrl": "https://www.vals.ai/benchmarks/corp_fin_v2",
      "paperTitle": "CorpFin v2",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "Credit-agreement understanding tasks",
      "format": "Accuracy score",
      "difficulty": "Professional finance document reasoning",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/valscorpfinv2",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/valscorpfinv2.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "valsMedCode",
      "name": "MedCode",
      "fullName": "Vals MedCode",
      "description": "Vals AI healthcare benchmark for whether models can support the medical billing process.",
      "paperUrl": "https://www.vals.ai/benchmarks/medcode",
      "paperTitle": "MedCode",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "Medical billing support tasks",
      "format": "Accuracy score",
      "difficulty": "Professional healthcare administration",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/valsmedcode",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/valsmedcode.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "valsMedScribe",
      "name": "MedScribe",
      "fullName": "Vals MedScribe",
      "description": "Vals AI healthcare benchmark for whether models can support doctors with administrative work.",
      "paperUrl": "https://www.vals.ai/benchmarks/medscribe",
      "paperTitle": "MedScribe",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "Medical administrative support tasks",
      "format": "Accuracy score",
      "difficulty": "Professional healthcare administration",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/valsmedscribe",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/valsmedscribe.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "valsMortgageTax",
      "name": "MortgageTax",
      "fullName": "Vals MortgageTax",
      "description": "Vals AI benchmark for mortgage and tax document reasoning, including semantic and numerical extraction task views.",
      "paperUrl": "https://www.vals.ai/benchmarks/mortgage_tax",
      "paperTitle": "MortgageTax",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "Mortgage and tax extraction tasks",
      "format": "Accuracy score",
      "difficulty": "Professional mortgage-tax document reasoning",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/valsmortgagetax",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/valsmortgagetax.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "valsProofBench",
      "name": "ProofBench",
      "fullName": "Vals ProofBench",
      "description": "Vals AI automated theorem-proving benchmark.",
      "paperUrl": "https://www.vals.ai/benchmarks/proof_bench",
      "paperTitle": "ProofBench",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "Automated theorem proving",
      "format": "Accuracy score",
      "difficulty": "Formal proof reasoning",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/valsproofbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/valsproofbench.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "valsLegalBench",
      "name": "LegalBench",
      "fullName": "Vals LegalBench",
      "description": "Vals AI legal benchmark with issue, rule, conclusion, interpretation, and rhetoric task views.",
      "paperUrl": "https://www.vals.ai/benchmarks/legal_bench",
      "paperTitle": "LegalBench",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "Legal reasoning task views",
      "format": "Accuracy score",
      "difficulty": "Professional legal reasoning",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/valslegalbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/valslegalbench.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "valsCaseLawV2",
      "name": "CaseLaw v2",
      "fullName": "Vals CaseLaw v2",
      "description": "Vals AI private question-answer benchmark over Canadian court cases.",
      "paperUrl": "https://www.vals.ai/benchmarks/case_law_v2",
      "paperTitle": "CaseLaw v2",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "Canadian case-law question answering",
      "format": "Accuracy score",
      "difficulty": "Professional legal retrieval and reasoning",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/valscaselawv2",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/valscaselawv2.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "deepSwe",
      "name": "DeepSWE",
      "fullName": "DeepSWE",
      "description": "A long-horizon software engineering benchmark from Datacurve for measuring frontier coding agents on original tasks drawn from active open-source repositories.",
      "paperUrl": "https://deepswe.datacurve.ai/blog",
      "paperTitle": "DeepSWE benchmark blog",
      "authors": "Datacurve AI",
      "year": "2026",
      "tasks": "113 software engineering tasks across 91 repositories and 5 languages",
      "format": "Pass@1 with confidence interval, cost, time, and token metadata",
      "difficulty": "Long-horizon software engineering",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/deepswe",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/deepswe.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "sweMarathon",
      "name": "SWE-Marathon",
      "fullName": "SWE-Marathon",
      "description": "A long-horizon software engineering benchmark from Abundant AI with multi-hour tasks spanning library reproductions, full-stack product clones, and ML engineering.",
      "paperUrl": "https://www.swe-marathon.org/",
      "paperTitle": "SWE-Marathon: Can Agents Autonomously Complete Ultra-Long-Horizon Software Work?",
      "authors": "Abundant AI and BenchFlow",
      "year": "2026",
      "tasks": "20 multi-hour software engineering tasks",
      "format": "Task resolution and trajectory review",
      "difficulty": "Ultra-long-horizon software engineering",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/swemarathon",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/swemarathon.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "exploitBench",
      "name": "ExploitBench",
      "fullName": "ExploitBench v8-bench",
      "description": "A cybersecurity benchmark for evaluating LLM agents on full-control V8 exploit synthesis using 16 measured exploit capability flags.",
      "paperUrl": "https://exploitbench.ai/",
      "paperTitle": "ExploitBench",
      "authors": "Seunghyun Lee, David Brumley, Carnegie Mellon University",
      "year": "2026",
      "tasks": "V8 exploit synthesis runs",
      "format": "Capability coverage percentage over 16 flags",
      "difficulty": "Browser exploitation and cybersecurity",
      "decimals": 0,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 7,
      "url": "https://benchlm.ai/benchmarks/exploitbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/exploitbench.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "aceCyberRangeSolved",
      "name": "ACE solved",
      "fullName": "ACE Cyber Range Challenges Solved",
      "description": "Number of advanced cyber-range challenges solved in the joint NIST CAISI and UK AISI preliminary evaluation.",
      "paperUrl": "https://www.nist.gov/news-events/news/2026/07/uk-aisi-caisi-preliminary-assessment-kimi-k3s-cyber-capabilities",
      "paperTitle": "UK AISI and CAISI Preliminary Assessment of Kimi K3's Cyber Capabilities",
      "authors": "NIST CAISI and UK AISI",
      "year": "2026",
      "tasks": "41 advanced cyber-range challenges",
      "format": "Challenges solved",
      "difficulty": "Advanced cyber operations",
      "decimals": 0,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/acecyberrangesolved",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/acecyberrangesolved.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "lastOnesCyberRangeSteps",
      "name": "The Last Ones steps",
      "fullName": "The Last Ones Average Progress",
      "description": "Average step reached on a 32-step long-horizon cyber range.",
      "paperUrl": "https://www.nist.gov/news-events/news/2026/07/uk-aisi-caisi-preliminary-assessment-kimi-k3s-cyber-capabilities",
      "paperTitle": "UK AISI and CAISI Preliminary Assessment of Kimi K3's Cyber Capabilities",
      "authors": "NIST CAISI and UK AISI",
      "year": "2026",
      "tasks": "32-step long-horizon cyber range",
      "format": "Average step reached",
      "difficulty": "Long-horizon cyber operations",
      "decimals": 0,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/lastonescyberrangesteps",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/lastonescyberrangesteps.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "lastOnesCyberRangeCompletion",
      "name": "The Last Ones completion",
      "fullName": "The Last Ones Cyber Range Completion Rate",
      "description": "Share of runs that completed the 32-step cyber range within the 100-million-token limit.",
      "paperUrl": "https://www.nist.gov/news-events/news/2026/07/uk-aisi-caisi-preliminary-assessment-kimi-k3s-cyber-capabilities",
      "paperTitle": "UK AISI and CAISI Preliminary Assessment of Kimi K3's Cyber Capabilities",
      "authors": "NIST CAISI and UK AISI",
      "year": "2026",
      "tasks": "10 long-horizon runs",
      "format": "Completion rate",
      "difficulty": "Long-horizon cyber operations",
      "decimals": 0,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 3,
      "url": "https://benchlm.ai/benchmarks/lastonescyberrangecompletion",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/lastonescyberrangecompletion.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "advancedCyberCompletionRateStandard",
      "name": "ACCR standard",
      "fullName": "Advanced Cyber Completion Rate — Standard Access",
      "description": "Share of approved advanced-cyber requests completed rather than refused under standard GPT-5.6 Sol safeguards.",
      "paperUrl": "https://openai.com/index/expanding-daybreak-as-the-cyber-defense-window-narrows/",
      "paperTitle": "Expanding Daybreak as the cyber defense window narrows",
      "authors": "OpenAI",
      "year": "2026",
      "tasks": "Internal advanced-cyber request set",
      "format": "Completion rate",
      "difficulty": "Cyber access and safeguards",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/advancedcybercompletionratestandard",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/advancedcybercompletionratestandard.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "advancedCyberCompletionRateBlue",
      "name": "ACCR Daybreak Blue",
      "fullName": "Advanced Cyber Completion Rate — Daybreak Blue",
      "description": "Share of approved advanced-cyber requests completed rather than refused by GPT-5.6 Sol under Daybreak Blue safeguards.",
      "paperUrl": "https://openai.com/index/expanding-daybreak-as-the-cyber-defense-window-narrows/",
      "paperTitle": "Expanding Daybreak as the cyber defense window narrows",
      "authors": "OpenAI",
      "year": "2026",
      "tasks": "Internal advanced-cyber request set",
      "format": "Completion rate",
      "difficulty": "Cyber access and safeguards",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/advancedcybercompletionrateblue",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/advancedcybercompletionrateblue.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "advancedCyberCompletionRateRed",
      "name": "ACCR Daybreak Red",
      "fullName": "Advanced Cyber Completion Rate — Daybreak Red",
      "description": "Share of approved advanced-cyber requests completed rather than refused by GPT-5.6 Cyber under Daybreak Red access.",
      "paperUrl": "https://openai.com/index/expanding-daybreak-as-the-cyber-defense-window-narrows/",
      "paperTitle": "Expanding Daybreak as the cyber defense window narrows",
      "authors": "OpenAI",
      "year": "2026",
      "tasks": "Internal advanced-cyber request set",
      "format": "Completion rate",
      "difficulty": "Cyber access and safeguards",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/advancedcybercompletionratered",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/advancedcybercompletionratered.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "secBenchPro",
      "name": "SEC-Bench Pro",
      "fullName": "SEC-Bench Pro",
      "description": "Cybersecurity benchmark for agentic vulnerability analysis and exploit-oriented security tasks.",
      "paperUrl": "https://openai.com/index/gpt-5-6/",
      "paperTitle": "Introducing GPT-5.6",
      "authors": "OpenAI",
      "year": "2026",
      "tasks": "Security engineering tasks",
      "format": "Success rate",
      "difficulty": "Advanced cybersecurity",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/secbenchpro",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/secbenchpro.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "frontierCyber",
      "name": "FrontierCyber",
      "fullName": "FrontierCyber",
      "description": "Independent evaluation of AI agents against vulnerable real-world systems in dynamic environments.",
      "paperUrl": "https://www.irregular.com/research/frontiercyber",
      "paperTitle": "FrontierCyber",
      "authors": "Irregular",
      "year": "2026",
      "tasks": "197 dynamic cyber tasks",
      "format": "Tasks solved",
      "difficulty": "Easy through elite cyber operations",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/frontiercyber",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/frontiercyber.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "cyScenarioBenchAverageSuccess",
      "name": "CyScenarioBench success",
      "fullName": "CyScenarioBench Average Success Rate",
      "description": "Average success rate across realistic, long-horizon cybersecurity scenarios.",
      "paperUrl": "https://www.irregular.com/research/assessing-gpt-5.6-sol",
      "paperTitle": "Assessing GPT-5.6 Sol",
      "authors": "Irregular",
      "year": "2026",
      "tasks": "11 cyber scenarios",
      "format": "Average success rate",
      "difficulty": "Long-horizon cybersecurity",
      "decimals": 0,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/cyscenariobenchaveragesuccess",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/cyscenariobenchaveragesuccess.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "cyScenarioBenchScenariosSolved",
      "name": "CyScenarioBench solved",
      "fullName": "CyScenarioBench Scenarios Ever Solved",
      "description": "Number of CyScenarioBench scenarios completed successfully in at least one run.",
      "paperUrl": "https://www.irregular.com/research/assessing-gpt-5.6-sol",
      "paperTitle": "Assessing GPT-5.6 Sol",
      "authors": "Irregular",
      "year": "2026",
      "tasks": "11 cyber scenarios",
      "format": "Scenarios solved at least once",
      "difficulty": "Long-horizon cybersecurity",
      "decimals": 0,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/cyscenariobenchscenariossolved",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/cyscenariobenchscenariossolved.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "atomicNetworkAttackSimulation",
      "name": "Atomic network attacks",
      "fullName": "Atomic Network Attack Simulation",
      "description": "Irregular's domain-level evaluation of network attack simulation capability.",
      "paperUrl": "https://www.irregular.com/research/assessing-gpt-5.6-sol",
      "paperTitle": "Assessing GPT-5.6 Sol",
      "authors": "Irregular",
      "year": "2026",
      "tasks": "Atomic cyber tasks",
      "format": "Domain average",
      "difficulty": "Network attack simulation",
      "decimals": 0,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/atomicnetworkattacksimulation",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/atomicnetworkattacksimulation.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "atomicVulnerabilityResearch",
      "name": "Atomic vulnerability research",
      "fullName": "Atomic Vulnerability Research and Exploitation",
      "description": "Irregular's domain-level evaluation of vulnerability research and exploitation capability.",
      "paperUrl": "https://www.irregular.com/research/assessing-gpt-5.6-sol",
      "paperTitle": "Assessing GPT-5.6 Sol",
      "authors": "Irregular",
      "year": "2026",
      "tasks": "Atomic cyber tasks",
      "format": "Domain average",
      "difficulty": "Vulnerability research and exploitation",
      "decimals": 0,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/atomicvulnerabilityresearch",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/atomicvulnerabilityresearch.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "atomicEvasion",
      "name": "Atomic evasion",
      "fullName": "Atomic Evasion",
      "description": "Irregular's domain-level evaluation of cybersecurity evasion capability.",
      "paperUrl": "https://www.irregular.com/research/assessing-gpt-5.6-sol",
      "paperTitle": "Assessing GPT-5.6 Sol",
      "authors": "Irregular",
      "year": "2026",
      "tasks": "Atomic cyber tasks",
      "format": "Domain average",
      "difficulty": "Cybersecurity evasion",
      "decimals": 0,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 2,
      "url": "https://benchlm.ai/benchmarks/atomicevasion",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/atomicevasion.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "sconePostCutoffSuccess",
      "name": "SCONE post-cutoff success",
      "fullName": "SCONE Post-Cutoff Exploit Success",
      "description": "Share of SCONE smart-contract vulnerabilities exploited on the 12-task post-cutoff set.",
      "paperUrl": "https://www.anthropic.com/research/exploit-evals",
      "paperTitle": "Evaluating frontier models on software exploitation",
      "authors": "Anthropic",
      "year": "2026",
      "tasks": "12 post-cutoff smart-contract vulnerabilities",
      "format": "Best@8 exploit success rate",
      "difficulty": "Smart-contract exploitation",
      "decimals": 0,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/sconepostcutoffsuccess",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/sconepostcutoffsuccess.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "sconePostCutoffRevenueUsdM",
      "name": "SCONE simulated revenue",
      "fullName": "SCONE Post-Cutoff Simulated Exploit Revenue",
      "description": "Simulated value captured across the SCONE post-cutoff smart-contract set.",
      "paperUrl": "https://www.anthropic.com/research/exploit-evals",
      "paperTitle": "Evaluating frontier models on software exploitation",
      "authors": "Anthropic",
      "year": "2026",
      "tasks": "12 post-cutoff smart-contract vulnerabilities",
      "format": "Simulated USD millions, Best@8",
      "difficulty": "Smart-contract exploitation",
      "decimals": 0,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/sconepostcutoffrevenueusdm",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/sconepostcutoffrevenueusdm.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "firefox147WorkingExploit",
      "name": "Firefox 147 exploits",
      "fullName": "Firefox 147 Working Exploit Rate",
      "description": "Share of patched Firefox 147 JavaScript-engine targets for which the model produced a working arbitrary-code-execution exploit.",
      "paperUrl": "https://www.anthropic.com/news/claude-fable-5-mythos-5",
      "paperTitle": "Claude Fable 5 and Claude Mythos 5",
      "authors": "Anthropic",
      "year": "2026",
      "tasks": "250 Firefox 147 vulnerability trials",
      "format": "Working arbitrary-code-execution rate",
      "difficulty": "Browser exploit development",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/firefox147workingexploit",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/firefox147workingexploit.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "anthropicOssFuzzAnyCrash",
      "name": "Anthropic OSS-Fuzz crash",
      "fullName": "Anthropic OSS-Fuzz Any-Crash Rate",
      "description": "Share of evaluated OSS-Fuzz entry points where the model produced at least a crash.",
      "paperUrl": "https://www.anthropic.com/news/claude-fable-5-mythos-5",
      "paperTitle": "Claude Fable 5 and Claude Mythos 5",
      "authors": "Anthropic",
      "year": "2026",
      "tasks": "Approximately 830 OSS-Fuzz entry points",
      "format": "Any-crash rate",
      "difficulty": "Vulnerability discovery",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/anthropicossfuzzanycrash",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/anthropicossfuzzanycrash.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "anthropicOssFuzzWritePrimitive",
      "name": "Anthropic OSS-Fuzz write primitive",
      "fullName": "Anthropic OSS-Fuzz Write-Primitive-or-Higher Rate",
      "description": "Share of evaluated OSS-Fuzz entry points where the model achieved a write primitive or stronger result.",
      "paperUrl": "https://www.anthropic.com/news/claude-fable-5-mythos-5",
      "paperTitle": "Claude Fable 5 and Claude Mythos 5",
      "authors": "Anthropic",
      "year": "2026",
      "tasks": "Approximately 830 OSS-Fuzz entry points",
      "format": "Write-primitive-or-higher rate",
      "difficulty": "Vulnerability exploitation",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 1,
      "url": "https://benchlm.ai/benchmarks/anthropicossfuzzwriteprimitive",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/anthropicossfuzzwriteprimitive.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "cveBenchZeroDayBlackBox",
      "name": "CVE-Bench zero-day",
      "fullName": "CVE-Bench v1 Zero-Day Black-Box Evaluation",
      "description": "OpenAI's black-box, no-source variant of CVE-Bench v1 across 40 critical vulnerabilities.",
      "paperUrl": "https://github.com/uiuc-kang-lab/cve-bench",
      "paperTitle": "CVE-Bench",
      "authors": "UIUC Kang Lab",
      "year": "2026",
      "tasks": "40 critical CVEs",
      "format": "Pass@1 over three rollouts",
      "difficulty": "Black-box vulnerability exploitation",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/cvebenchzerodayblackbox",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/cvebenchzerodayblackbox.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "gbaEval",
      "name": "GBA-Eval",
      "fullName": "GBA-Eval",
      "description": "An agentic coding benchmark that asks models to build a Game Boy Advance emulator from scratch and grades emulator behavior against procedural, audio, and gameplay tests.",
      "paperUrl": "https://gbaeval.com/",
      "paperTitle": "GBA-Eval",
      "authors": "Stephen Yang",
      "year": "2026",
      "tasks": "27 emulator test cases",
      "format": "Overall emulator score",
      "difficulty": "Long-horizon systems programming",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/gbaeval",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/gbaeval.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "caisTextLeaderboard",
      "name": "CAIS Text Leaderboard",
      "fullName": "CAIS AI Dashboard Text Capabilities Index",
      "description": "A Center for AI Safety dashboard view summarizing text capabilities across HLE, ARC-AGI-2, SWE-Bench Pro, and TextQuests.",
      "paperUrl": "https://dashboard.safe.ai/",
      "paperTitle": "CAIS AI Dashboard",
      "authors": "Center for AI Safety",
      "year": "2025",
      "tasks": "HLE, ARC-AGI-2, SWE-Bench Pro, and TextQuests",
      "format": "Average component score",
      "difficulty": "Composite frontier text capability",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/caistextleaderboard",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/caistextleaderboard.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "voxelbench",
      "name": "VoxelBench Text",
      "fullName": "VoxelBench Text-Prompt Leaderboard",
      "description": "A live human-preference benchmark where language models turn text prompts into voxel structures and voters compare anonymous builds from the same prompt.",
      "paperUrl": "https://voxelbench.ai/leaderboard",
      "paperTitle": "VoxelBench leaderboard",
      "authors": "VoxelBench",
      "year": "2025",
      "tasks": "Live text prompts for 3D voxel construction",
      "format": "Glicko-2 rating from blind pairwise votes",
      "difficulty": "3D spatial construction and visual quality",
      "decimals": 0,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/voxelbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/voxelbench.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "voxelbench-image",
      "name": "VoxelBench Image",
      "fullName": "VoxelBench Image-Prompt Leaderboard",
      "description": "A live human-preference benchmark where multimodal models build voxel structures from image references and voters compare anonymous results produced from the same prompt.",
      "paperUrl": "https://voxelbench.ai/leaderboard",
      "paperTitle": "VoxelBench leaderboard",
      "authors": "VoxelBench",
      "year": "2025",
      "tasks": "Live image-reference prompts for 3D voxel construction",
      "format": "Glicko-2 rating from blind pairwise votes",
      "difficulty": "Visual grounding, 3D construction, and aesthetic quality",
      "decimals": 0,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/voxelbench-image",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/voxelbench-image.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "weirdMl",
      "name": "WeirdML",
      "fullName": "WeirdML v2",
      "description": "A machine-learning engineering benchmark that tests whether LLMs can train models on novel datasets, write PyTorch code, and improve through iterative feedback.",
      "paperUrl": "https://htihle.github.io/weirdml.html",
      "paperTitle": "WeirdML",
      "authors": "Havard Tveit Ihle",
      "year": "2026",
      "tasks": "17 novel ML engineering tasks",
      "format": "Average accuracy across tasks",
      "difficulty": "Novel dataset modeling and iterative debugging",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/weirdml",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/weirdml.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "aleBench",
      "name": "ALE-Bench",
      "fullName": "Agents Last Exam",
      "description": "A benchmark for agentic professional workflows with verifiable success criteria, reporting pass rates and partial scores for model plus agent-harness rows.",
      "paperUrl": "https://agents-last-exam.org/leaderboard",
      "paperTitle": "Agents Last Exam",
      "authors": "UC Berkeley RDI",
      "year": "2026",
      "tasks": "152 ALE-V1 professional workflow tasks across 13 top-level domains",
      "format": "Pass rate, partial-credit score, cost, token, and duration metadata",
      "difficulty": "Real-world agentic workflows",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/alebench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/alebench.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "runescapeBench",
      "name": "RuneScape-Bench",
      "fullName": "RuneBench / runescape-bench",
      "description": "An agentic coding benchmark where models use a TypeScript SDK to play a RuneScape-like environment and optimize skill-training performance.",
      "paperUrl": "https://maxbittker.github.io/runebench/",
      "paperTitle": "RuneBench",
      "authors": "Max Bittker",
      "year": "2026",
      "tasks": "16 RuneScape skill-training tasks",
      "format": "Average log XP-rate score",
      "difficulty": "Agentic gameplay automation",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/runescapebench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/runescapebench.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "tolokaArena",
      "name": "Toloka Arena",
      "fullName": "Toloka Arena",
      "description": "An independent agentic-intelligence evaluation from Toloka using private simulated workflows and a pass^5 metric.",
      "paperUrl": "https://toloka.ai/arena",
      "paperTitle": "Toloka Arena",
      "authors": "Toloka",
      "year": "2026",
      "tasks": "Private simulated enterprise workflows",
      "format": "pass^5 arena score",
      "difficulty": "Agentic workflow reliability",
      "decimals": 1,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/tolokaarena",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/tolokaarena.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "valsSweBench",
      "name": "Vals SWE-bench mirror",
      "fullName": "Vals-hosted SWE-bench mirror",
      "description": "Vals AI hosted SWE-bench view for solving production software engineering tasks.",
      "paperUrl": "https://www.vals.ai/benchmarks/swebench",
      "paperTitle": "Vals SWE-bench",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "Software engineering issue-resolution tasks",
      "format": "Accuracy score",
      "difficulty": "Production software engineering",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/valsswebench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/valsswebench.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "valsTerminalBench2",
      "name": "Vals Terminal-Bench 2.0 mirror",
      "fullName": "Vals-hosted Terminal-Bench 2.0 mirror",
      "description": "Vals AI hosted Terminal-Bench 2.0 view with easy, medium, and hard task splits.",
      "paperUrl": "https://www.vals.ai/benchmarks/terminal-bench-2",
      "paperTitle": "Vals Terminal-Bench 2.0",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "Terminal task difficulty splits",
      "format": "Accuracy score",
      "difficulty": "Terminal-based agent execution",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/valsterminalbench2",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/valsterminalbench2.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "valsLiveCodeBench",
      "name": "Vals LiveCodeBench mirror",
      "fullName": "Vals-hosted LiveCodeBench mirror",
      "description": "Vals AI implementation of LiveCodeBench with easy, medium, and hard task splits.",
      "paperUrl": "https://www.vals.ai/benchmarks/lcb",
      "paperTitle": "Vals LiveCodeBench",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "Coding problem difficulty splits",
      "format": "Accuracy score",
      "difficulty": "Contamination-resistant coding problems",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/valslivecodebench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/valslivecodebench.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "valsGpqaDiamond",
      "name": "Vals GPQA Diamond mirror",
      "fullName": "Vals-hosted GPQA Diamond mirror",
      "description": "Vals AI hosted GPQA Diamond view with few-shot and zero-shot chain-of-thought task splits.",
      "paperUrl": "https://www.vals.ai/benchmarks/gpqa",
      "paperTitle": "Vals GPQA Diamond",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "GPQA Diamond task splits",
      "format": "Accuracy score",
      "difficulty": "Graduate science reasoning",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/valsgpqadiamond",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/valsgpqadiamond.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "valsMmluPro",
      "name": "Vals MMLU-Pro mirror",
      "fullName": "Vals-hosted MMLU-Pro mirror",
      "description": "Vals AI hosted MMLU-Pro view with subject-level task splits.",
      "paperUrl": "https://www.vals.ai/benchmarks/mmlu_pro",
      "paperTitle": "Vals MMLU-Pro",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "MMLU-Pro subject splits",
      "format": "Accuracy score",
      "difficulty": "Professional academic reasoning",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/valsmmlupro",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/valsmmlupro.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "emb",
      "name": "EMB",
      "fullName": "Vals EMB",
      "description": "Evaluating agents on Excel-based financial modeling tasks",
      "paperUrl": "https://www.vals.ai/benchmarks/emb",
      "paperTitle": "EMB",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "Excel-based financial modeling tasks",
      "format": "Accuracy score",
      "difficulty": "Professional finance modeling",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/emb",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/emb.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "cyber",
      "name": "CyberBench",
      "fullName": "Vals CyberBench",
      "description": "Can autonomous agents craft PoC inputs that trigger OSS-Fuzz vulnerabilities—and stop crashing after the fix?",
      "paperUrl": "https://www.vals.ai/benchmarks/cyber",
      "paperTitle": "CyberBench",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "OSS-Fuzz PoC and patch-verification tasks",
      "format": "Accuracy score",
      "difficulty": "Autonomous cybersecurity exploit reproduction",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/cyber",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/cyber.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "taxEvalV2",
      "name": "TaxEval v2",
      "fullName": "Vals TaxEval v2",
      "description": "A Vals-created set of questions and responses to tax questions",
      "paperUrl": "https://www.vals.ai/benchmarks/tax_eval_v2",
      "paperTitle": "TaxEval v2",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "Tax question answering and response evaluation",
      "format": "Accuracy score",
      "difficulty": "Professional tax reasoning",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/taxevalv2",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/taxevalv2.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "hlab",
      "name": "Harvey's Legal Agent Benchmark",
      "fullName": "Vals Harvey's Legal Agent Benchmark",
      "description": "Tests an agent's ability to complete legal work using documents, spreadsheets, presentations, and file-system tools",
      "paperUrl": "https://www.vals.ai/benchmarks/hlab",
      "paperTitle": "Harvey's Legal Agent Benchmark",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "Legal agent work across documents, spreadsheets, presentations, and files",
      "format": "Accuracy score",
      "difficulty": "Professional legal workflow automation",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/hlab",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/hlab.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "valsTerminalBench21",
      "name": "Terminal-Bench 2.1",
      "fullName": "Vals Terminal-Bench 2.1",
      "description": "State-of-the-art set of difficult terminal-based tasks",
      "paperUrl": "https://www.vals.ai/benchmarks/terminal-bench-2-1",
      "paperTitle": "Terminal-Bench 2.1",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "Terminal-based task execution",
      "format": "Accuracy score",
      "difficulty": "Frontier terminal-agent execution",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": "https://benchlm.ai/benchmarks/terminal-bench-4",
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/valsterminalbench21",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/valsterminalbench21.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "codeMigration",
      "name": "Code Migration",
      "fullName": "Vals Code Migration",
      "description": "Can language models reimplement real-world programs in another language?",
      "paperUrl": "https://www.vals.ai/benchmarks/code-migration",
      "paperTitle": "Code Migration",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "Real-world program reimplementation in another language",
      "format": "Accuracy score",
      "difficulty": "Production code migration",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/codemigration",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/codemigration.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "legalResearchBench",
      "name": "Legal Research Bench",
      "fullName": "Vals Legal Research Bench",
      "description": "Evaluating agents on legal research tasks across diverse areas of US law",
      "paperUrl": "https://www.vals.ai/benchmarks/legal_research",
      "paperTitle": "Legal Research Bench",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "US-law legal research tasks",
      "format": "Accuracy score",
      "difficulty": "Professional legal research",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/legalresearchbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/legalresearchbench.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "valsMedQa",
      "name": "MedQA",
      "fullName": "Vals MedQA",
      "description": "Evaluating language model bias in medical questions.",
      "paperUrl": "https://www.vals.ai/benchmarks/medqa",
      "paperTitle": "MedQA",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "Medical question answering",
      "format": "Accuracy score",
      "difficulty": "Medical knowledge and bias evaluation",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/valsmedqa",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/valsmedqa.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "valsAime",
      "name": "AIME",
      "fullName": "Vals AIME",
      "description": "Challenging national math exam given to top high-school students",
      "paperUrl": "https://www.vals.ai/benchmarks/aime",
      "paperTitle": "AIME",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "AIME math problems",
      "format": "Accuracy score",
      "difficulty": "Competition math",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/valsaime",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/valsaime.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "valsMath500",
      "name": "MATH 500",
      "fullName": "Vals MATH 500",
      "description": "Academic math benchmark on probability, algebra, and trigonometry",
      "paperUrl": "https://www.vals.ai/benchmarks/math500",
      "paperTitle": "MATH 500",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "MATH 500 academic math problems",
      "format": "Accuracy score",
      "difficulty": "Advanced academic math",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/valsmath500",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/valsmath500.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "valsMgsm",
      "name": "MGSM",
      "fullName": "Vals MGSM",
      "description": "A multilingual benchmark for mathematical questions.",
      "paperUrl": "https://www.vals.ai/benchmarks/mgsm",
      "paperTitle": "MGSM",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "Multilingual grade-school math questions",
      "format": "Accuracy score",
      "difficulty": "Multilingual mathematical reasoning",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/valsmgsm",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/valsmgsm.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "valsMmmu",
      "name": "MMMU",
      "fullName": "Vals MMMU",
      "description": "Multimodal Multi-task Benchmark",
      "paperUrl": "https://www.vals.ai/benchmarks/mmmu",
      "paperTitle": "MMMU",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "Multimodal academic task suite",
      "format": "Accuracy score",
      "difficulty": "Multimodal college-level reasoning",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/valsmmmu",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/valsmmmu.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "sage",
      "name": "SAGE",
      "fullName": "Vals SAGE",
      "description": "Student Assessment with Generative Evaluation",
      "paperUrl": "https://www.vals.ai/benchmarks/sage",
      "paperTitle": "SAGE",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "Student assessment with generative evaluation",
      "format": "Accuracy score",
      "difficulty": "Education assessment reasoning",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/sage",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/sage.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "valsIoi",
      "name": "IOI",
      "fullName": "Vals IOI",
      "description": "Based on the International Olympiad in Informatics",
      "paperUrl": "https://www.vals.ai/benchmarks/ioi",
      "paperTitle": "IOI",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "International Olympiad in Informatics-style programming tasks",
      "format": "Accuracy score",
      "difficulty": "Olympiad programming",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/valsioi",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/valsioi.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "valsProgramBench",
      "name": "ProgramBench",
      "fullName": "Vals ProgramBench",
      "description": "Can language models rebuild programs from scratch?",
      "paperUrl": "https://www.vals.ai/benchmarks/programbench",
      "paperTitle": "ProgramBench",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "Program reconstruction tasks",
      "format": "Accuracy score",
      "difficulty": "Cleanroom software engineering",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/valsprogrambench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/valsprogrambench.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "skillsBench",
      "name": "SkillsBench",
      "fullName": "Vals SkillsBench",
      "description": "How important are skills for agents?",
      "paperUrl": "https://www.vals.ai/benchmarks/skillsbench",
      "paperTitle": "SkillsBench",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "Agent skill-importance tasks",
      "format": "Accuracy score",
      "difficulty": "Agent skill evaluation",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/skillsbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/skillsbench.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "pokerAgent",
      "name": "Agent Poker Bench",
      "fullName": "Vals Agent Poker Bench",
      "description": "Which model can make the most money playing poker?",
      "paperUrl": "https://www.vals.ai/benchmarks/poker_agent",
      "paperTitle": "Agent Poker Bench",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "Poker-playing agent trials",
      "format": "Accuracy score",
      "difficulty": "Strategic game-agent decision making",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/pokeragent",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/pokeragent.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "publicBenefitsBench",
      "name": "Public Benefits Bench v1.1",
      "fullName": "Vals Public Benefits Bench v1.1",
      "description": "Can AI help people navigate SNAP benefits?",
      "paperUrl": "https://www.vals.ai/benchmarks/public-benefits-bench",
      "paperTitle": "Public Benefits Bench v1.1",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "SNAP public-benefits navigation tasks",
      "format": "Accuracy score",
      "difficulty": "Public-benefits policy navigation",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/publicbenefitsbench",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/publicbenefitsbench.md"
    },
    {
      "category": "external",
      "categoryLabel": "external",
      "benchmarkKey": "publicBenefitsBenchV1",
      "name": "Public Benefits Bench v1",
      "fullName": "Vals Public Benefits Bench v1",
      "description": "Can AI help people navigate SNAP benefits?",
      "paperUrl": "https://www.vals.ai/benchmarks/public-benefits-bench-v1",
      "paperTitle": "Public Benefits Bench v1",
      "authors": "Vals AI",
      "year": "2026",
      "tasks": "SNAP public-benefits navigation tasks",
      "format": "Accuracy score",
      "difficulty": "Public-benefits policy navigation",
      "decimals": 2,
      "successorKey": null,
      "successorUrl": null,
      "weight": null,
      "displayableScoreCount": 0,
      "url": "https://benchlm.ai/benchmarks/publicbenefitsbenchv1",
      "markdownUrl": "https://benchlm.ai/md/benchmarks/publicbenefitsbenchv1.md"
    }
  ]
}
