{
  "schemaVersion": "1.0",
  "name": "BenchLM models",
  "description": "Stable model metadata, rankings, benchmark scores, and coverage fields for BenchLM model pages.",
  "canonicalUrl": "https://benchlm.ai/data/models.json",
  "generatedAt": "2026-09-02T00:50:21.010Z",
  "sourceLastUpdated": "September 1, 2026",
  "sourceFiles": [
    "src/data/benchmarks.json",
    "src/data/provenance.js",
    "src/data/modelReleaseMetadata.js"
  ],
  "counts": {
    "totalModels": 406,
    "canonicalModels": 302,
    "rankingEligibleModels": 225
  },
  "items": [
    {
      "slug": "claude-opus-5",
      "canonicalModelKey": "claude-opus-5",
      "model": "Claude Opus 5",
      "creator": "Anthropic",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": null,
      "contextWindowTokens": 0,
      "displayScore": 82.34,
      "provisionalDisplayScore": 84,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 79.1,
        "upper": 85.59
      },
      "rankingEligible": true,
      "overallRank": 3,
      "url": "https://benchlm.ai/models/claude-opus-5",
      "markdownUrl": "https://benchlm.ai/md/models/claude-opus-5.md",
      "id": 295,
      "releaseDate": "2026-07-24",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "claude-opus-5",
        "familyName": "Claude Opus 5",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "claude-opus-5",
        "relatedModelKeys": [
          "claude-opus-4-8",
          "claude-fable-5",
          "claude-sonnet-5"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "claude-opus-4-8"
      },
      "scores": {
        "displayScore": 84,
        "overallScore": 84,
        "rawOverallScore": 84,
        "verifiedDisplayScore": 84,
        "displayCategoryScores": {
          "agentic": 84.6,
          "coding": 83,
          "reasoning": 83.5,
          "multimodalGrounded": 86.4,
          "knowledge": 97,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 84.2,
          "coding": 83,
          "reasoning": 83.5,
          "multimodalGrounded": 86.4,
          "knowledge": 97,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 3,
        "categoryRanks": {
          "agentic": 1,
          "coding": 5,
          "multimodalGrounded": 4,
          "knowledge": 1
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 59,
        "verifiedBenchmarkCount": 59,
        "rankableBenchmarkCount": 59,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 3
      },
      "benchmarks": {
        "agentic": {
          "frontierBench": 42.7,
          "browseComp": 90.8,
          "hleWithTools": 64.7,
          "deepSearchQa": 95,
          "draco": 88.6,
          "multiAgentBrowseCompPrerelease": 93.6,
          "osWorld2": 70.6,
          "mcpAtlas": 85.8,
          "mcpAtlasClaimCoverage": 89.1,
          "legalAgentBenchAllPass": 23.58,
          "legalAgentBenchCriterionPass": 93.74,
          "legalAgentBenchHeldoutAllPass": 11.7,
          "legalAgentBenchHeldoutCriterionPass": 94.1,
          "gdpvalAa": 1862,
          "toolathlonVerified": 80.6,
          "toolathlonVerifiedPass3": 87,
          "toolathlonVerifiedPass3All": 73.1,
          "toolathlonVerifiedAvgTurns": 23.5,
          "automationBench": 26,
          "aaAgenticIndex": 59.17,
          "gdpvalAaNormalized": 66.2,
          "aaTau3Banking": 42.1,
          "aaBriefcaseElo": 1720,
          "aaTerminalBench21": 89.1,
          "aaEnterpriseOpsGym": 47.5,
          "aaHarveyLab": 93.5
        },
        "coding": {
          "sweVerified": 96,
          "swePro": 79.2,
          "sweMultilingual": 89.5,
          "sweMultimodal": 59.4,
          "deepSwe": 68.8,
          "frontierCode": 53.4,
          "frontierCode11Extended": 63.6,
          "programBenchEpisode1": 83,
          "programBench": 93,
          "cursorBench32": 70,
          "vulcanBench": 87,
          "aaCodingIndex": 77.98,
          "aaSciCode": 55.7,
          "vulcanCiiV1": 96.4
        },
        "reasoning": {
          "arcAgi1": 97.5,
          "arcAgi2": 90.4,
          "arcAgi3": 30.16,
          "lcr": 75.7,
          "critpt": 29.1
        },
        "multimodalGrounded": {
          "chartography": 29.6,
          "chartographyWithTools": 83,
          "benchCadVision2Code": 0.366,
          "benchCadVision2CodeWithTools": 0.821,
          "gdpPdf": 83.4,
          "gdpPdfWithTools": 85.5,
          "officeQa": 78.1,
          "officeQaPro": 66.9,
          "aaMmmuPro": 84.7,
          "designArenaWebsite": 1327
        },
        "knowledge": {
          "hle": 64.7,
          "hleNoTools": 56.3,
          "healthBench": 67.1,
          "healthBenchLengthAdjusted": 57.8,
          "healthBenchProfessional": 59.8,
          "healthBenchProfessionalRaw": 73.4,
          "bioMysteryBenchHumanSolvable": 90.1,
          "bioMysteryBenchHumanDifficult": 49.4,
          "spatialBenchVerified": 72.5,
          "singleCellBench": 60.6,
          "proteinGymHard": 47.7,
          "proteinDesign": 42.5,
          "organicChemistryV2": 61.6,
          "protocolsTroubleshooting": 61.1,
          "protocolsUnderstanding": 78.4,
          "artificialAnalysis": 63.05,
          "aaGpqaDiamond": 93.2,
          "aaHle": 54.9,
          "aaOmniscienceIndex": 37.1,
          "omniscienceAccuracy": 60.9,
          "omniscienceHallucinationRate": 60.8
        },
        "multilingual": {
          "gmmlu": 92.5,
          "milu": 92.1,
          "include": 89.8
        },
        "instructionFollowing": {},
        "math": {
          "imo2026": 42,
          "riemannBench": 60,
          "riemannBenchWithTools": 79,
          "arxivMathJune2026": 90.8,
          "arxivMathJune2026WithTools": 91.3
        }
      }
    },
    {
      "slug": "kimi-k3",
      "canonicalModelKey": "kimi-3",
      "model": "Kimi K3",
      "creator": "Moonshot AI",
      "sourceType": "Pending",
      "reasoningType": "Reasoning",
      "contextWindow": "1.05M",
      "contextWindowTokens": 1050000,
      "displayScore": 79.92,
      "provisionalDisplayScore": 82,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 77.66,
        "upper": 82.19
      },
      "rankingEligible": true,
      "overallRank": 5,
      "url": "https://benchlm.ai/models/kimi-k3",
      "markdownUrl": "https://benchlm.ai/md/models/kimi-k3.md",
      "id": 282,
      "releaseDate": "2026-07-16",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "kimi-3",
        "familyName": "Kimi K3",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "kimi-3",
        "relatedModelKeys": [
          "kimi-k2-7-code",
          "kimi-2-6",
          "kimi-k2-5"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "kimi-2-6"
      },
      "scores": {
        "displayScore": 82,
        "overallScore": 82,
        "rawOverallScore": 81,
        "verifiedDisplayScore": 82,
        "displayCategoryScores": {
          "agentic": 94.5,
          "coding": 63,
          "reasoning": null,
          "multimodalGrounded": 88.4,
          "knowledge": 83.5,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 94.5,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": 88.4,
          "knowledge": 83.5,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 5,
        "categoryRanks": {
          "agentic": 5,
          "coding": 4,
          "multimodalGrounded": 1,
          "knowledge": 10
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 30,
        "verifiedBenchmarkCount": 30,
        "rankableBenchmarkCount": 9,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 88.3,
          "browseComp": 91.2,
          "deepSearchQa": 95,
          "toolathlonVerified": 73.2,
          "mcpAtlas": 84.2,
          "automationBench": 30.8,
          "jobBench": 52.9,
          "apexAgents": 37.6,
          "spreadsheetBench2": 34.8,
          "deckBench": 73.5,
          "aaAgenticIndex": 54.26,
          "gdpvalAaNormalized": 58.4,
          "gdpvalAa": 1668,
          "aaBriefcaseElo": 1517,
          "aaAutomationBench": 52.7,
          "aaEnterpriseOpsGym": 45.3,
          "aaHarveyLab": 94.6,
          "aaTau3Banking": 46,
          "aaTerminalBench21": 85,
          "apexAgentsAa": 41.3,
          "aaItbench": 47.7
        },
        "coding": {
          "deepSwe": 67.5,
          "cursorBench32": 60.8,
          "frontierSwe": 81.2,
          "programBench": 77.8,
          "kimiCodeBenchV2": 72.9,
          "sweMarathon": 42,
          "postTrainBench": 36.6,
          "mlsBenchLite": 48.3,
          "aaCodingIndex": 76.24,
          "aaSciCode": 58.7,
          "vulcanBench": 73.7,
          "openHarmonyBench": 57.3
        },
        "reasoning": {
          "lcr": 82.7,
          "critpt": 23.4
        },
        "multimodalGrounded": {
          "officeQaPro": 63.3,
          "mmmuPro": 81.6,
          "mmmuProPython": 83.4,
          "charxivNoTools": 84.8,
          "charxiv": 91.3,
          "mathVision": 94.3,
          "mathVisionPython": 97.8,
          "babyVisionPython": 85.7,
          "zeroBench": 23,
          "zeroBenchPython": 41,
          "worldVqaForceAnswer": 51,
          "omniDocBench": 91.1,
          "perceptionBench": 58.5,
          "aaMmmuPro": 80.5,
          "designArenaWebsite": 1363
        },
        "knowledge": {
          "gpqa": 93.5,
          "gpqaDiamond": 93.5,
          "hle": 56,
          "hleNoTools": 43.5,
          "artificialAnalysis": 59.7,
          "aaGpqaDiamond": 93.5,
          "aaHle": 46.9,
          "aaOmniscienceIndex": 19.7,
          "omniscienceAccuracy": 47.6,
          "omniscienceHallucinationRate": 53.2,
          "aaOpennessIndex": 38.9
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {},
        "external": {
          "exploitBench": 32,
          "aceCyberRangeSolved": 0,
          "lastOnesCyberRangeSteps": 17,
          "lastOnesCyberRangeCompletion": 10
        }
      }
    },
    {
      "slug": "gpt-5-6-sol",
      "canonicalModelKey": "gpt-5-6-sol",
      "model": "GPT-5.6 Sol",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1.05M",
      "contextWindowTokens": 1050000,
      "displayScore": 81.69,
      "provisionalDisplayScore": 80,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 78.87,
        "upper": 84.5
      },
      "rankingEligible": true,
      "overallRank": 4,
      "url": "https://benchlm.ai/models/gpt-5-6-sol",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-5-6-sol.md",
      "id": 263,
      "releaseDate": "2026-07-09",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-5-6",
        "familyName": "GPT-5.6",
        "variantType": "sol",
        "snapshotLabel": "sol",
        "baseFamilyModelKey": "gpt-5-6-sol",
        "relatedModelKeys": [
          "gpt-5-6-terra",
          "gpt-5-6-luna",
          "gpt-5-6-cyber"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "gpt-5-5"
      },
      "scores": {
        "displayScore": 80,
        "overallScore": 80,
        "rawOverallScore": 80,
        "verifiedDisplayScore": 80,
        "displayCategoryScores": {
          "agentic": 94.5,
          "coding": 55.3,
          "reasoning": 85.2,
          "multimodalGrounded": 83.2,
          "knowledge": 85.9,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 97
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 94.5,
          "coding": 53.5,
          "reasoning": 85.2,
          "multimodalGrounded": 83.2,
          "knowledge": 85.9,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 97
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": false,
        "overallRank": 4,
        "categoryRanks": {
          "agentic": 10,
          "coding": 3,
          "multimodalGrounded": 5,
          "knowledge": 5
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 23,
        "verifiedBenchmarkCount": 23,
        "rankableBenchmarkCount": 23,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 3
      },
      "benchmarks": {
        "agentic": {
          "frontierBench": 34.59,
          "terminalBench2": 91.9,
          "browseComp": 92.2,
          "osWorld2": 62.6,
          "cyberGym": 84.5,
          "exploitGym": 33.7,
          "toolathlon": 58,
          "aaAgenticIndex": 57.78,
          "tau2Bench": 85.1,
          "gdpvalAaNormalized": 60.5,
          "gdpvalAa": 1735,
          "aaBriefcaseElo": 1494,
          "aaItbench": 56.2,
          "aaTau3Banking": 44.3,
          "aaAutomationBench": 51.2,
          "aaHarveyLab": 87.2,
          "terminalBenchHard": 65.9,
          "aaTerminalBench21": 88,
          "aaEnterpriseOpsGym": 42.9
        },
        "coding": {
          "swePro": 64.6,
          "terminalBench2": 91.9,
          "deepSwe": 72.7,
          "frontierCode11Extended": 60.6,
          "cursorBench32": 67.2,
          "vulcanBench": 87,
          "aaCodingIndex": 77.39,
          "aaSciCode": 56.1,
          "vulcanCiiV1": 86.5
        },
        "reasoning": {
          "arcAgi2": 92.5,
          "arcAgi3": 7.78,
          "geneBenchPro": 28.7,
          "lcr": 77.7,
          "critpt": 32.3
        },
        "multimodalGrounded": {
          "mmmuPro": 83,
          "mmmuProPython": 84.6,
          "aaMmmuPro": 83.4
        },
        "knowledge": {
          "gpqa": 94.6,
          "gpqaDiamond": 94.6,
          "healthBenchProfessional": 60.5,
          "healthBenchHard": 33.1,
          "artificialAnalysis": 58.89,
          "aaGpqaDiamond": 94.1,
          "aaHle": 49.5,
          "aaOmniscienceIndex": 22,
          "omniscienceAccuracy": 59.4,
          "omniscienceHallucinationRate": 92.2
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 72.7
        },
        "math": {
          "frontierMath": 89,
          "frontierMathV2Tiers13": 89,
          "frontierMathV2Tier4": 83
        },
        "external": {
          "exploitBench": 73.5,
          "secBenchPro": 71.2,
          "lastOnesCyberRangeCompletion": 70,
          "advancedCyberCompletionRateStandard": 1.5,
          "advancedCyberCompletionRateBlue": 2,
          "frontierCyber": 9.6,
          "cyScenarioBenchAverageSuccess": 28,
          "cyScenarioBenchScenariosSolved": 7,
          "atomicNetworkAttackSimulation": 98,
          "atomicVulnerabilityResearch": 91,
          "atomicEvasion": 56
        }
      }
    },
    {
      "slug": "claude-fable-5-1",
      "canonicalModelKey": "claude-fable-5-1",
      "model": "Claude Fable 5.1",
      "creator": "Anthropic",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 82.74,
      "provisionalDisplayScore": 79,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 71.23,
        "upper": 94.26
      },
      "rankingEligible": true,
      "overallRank": 1,
      "url": "https://benchlm.ai/models/claude-fable-5-1",
      "markdownUrl": "https://benchlm.ai/md/models/claude-fable-5-1.md",
      "id": 416,
      "releaseDate": "2026-09-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "claude-fable",
        "familyName": "Claude Fable",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "claude-fable-5-1",
        "relatedModelKeys": [
          "claude-fable-5",
          "claude-mythos-5-1"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "claude-fable-5"
      },
      "scores": {
        "displayScore": 79,
        "overallScore": 79,
        "rawOverallScore": 79,
        "verifiedDisplayScore": 79,
        "displayCategoryScores": {
          "agentic": 85,
          "coding": 80.3,
          "reasoning": 83.1,
          "multimodalGrounded": null,
          "knowledge": 97,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": 80.3,
          "reasoning": 83.1,
          "multimodalGrounded": null,
          "knowledge": 97,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": false,
        "overallRank": 1,
        "categoryRanks": {
          "agentic": 2,
          "coding": 2,
          "knowledge": 2
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 14,
        "verifiedBenchmarkCount": 14,
        "rankableBenchmarkCount": 14,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "terminalBench4": 55.8,
          "terminalBenchScience": 52.6,
          "osWorld2": 41.7,
          "automationBench": 31.4,
          "toolathlonVerified": 77.8,
          "toolathlonVerifiedPass3": 81.5,
          "toolathlonVerifiedPass3All": 73.1,
          "toolathlonVerifiedAvgTurns": 23.7,
          "aaAgenticIndex": 61.34,
          "gdpvalAaNormalized": 67.7,
          "gdpvalAa": 1853,
          "aaBriefcaseElo": 1694,
          "aaHarveyLab": 93,
          "aaTau3Banking": 47.2,
          "aaTerminalBench21": 91.4
        },
        "coding": {
          "swePro": 81.2,
          "sweMultilingual": 89.1,
          "sweMultimodal": 54.7,
          "deepSwe": 67.4,
          "programBench": 87.6,
          "cursorBench32": 73.4,
          "aaCodingIndex": 81.6,
          "aaSciCode": 62
        },
        "reasoning": {
          "arcAgi1": 97.5,
          "arcAgi2": 90,
          "lcr": 80,
          "critpt": 29.7
        },
        "multimodalGrounded": {},
        "knowledge": {
          "hle": 65,
          "hleNoTools": 60.9,
          "artificialAnalysis": 65.65,
          "aaGpqaDiamond": 93.7,
          "aaHle": 59.1,
          "aaOmniscienceIndex": 43.5,
          "omniscienceAccuracy": 67.2,
          "omniscienceHallucinationRate": 72.6
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "claude-fable",
      "canonicalModelKey": "claude-fable-5",
      "model": "Claude Fable 5",
      "creator": "Anthropic",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M+",
      "contextWindowTokens": 1000000,
      "displayScore": 82.49,
      "provisionalDisplayScore": 78,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 80.26,
        "upper": 84.73
      },
      "rankingEligible": true,
      "overallRank": 2,
      "url": "https://benchlm.ai/models/claude-fable",
      "markdownUrl": "https://benchlm.ai/md/models/claude-fable.md",
      "id": 257,
      "releaseDate": "2026-06-09",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "claude-fable",
        "familyName": "Claude Fable",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "claude-fable-5",
        "relatedModelKeys": [
          "claude-mythos-5",
          "claude-fable-5-1"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 78,
        "overallScore": 78,
        "rawOverallScore": 78,
        "verifiedDisplayScore": 78,
        "displayCategoryScores": {
          "agentic": 92.7,
          "coding": 83,
          "reasoning": null,
          "multimodalGrounded": 61.3,
          "knowledge": 67,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 92.7,
          "coding": 83,
          "reasoning": null,
          "multimodalGrounded": 61.3,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": false,
        "overallRank": 2,
        "categoryRanks": {
          "agentic": 3,
          "coding": 1,
          "knowledge": 29
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 11,
        "verifiedBenchmarkCount": 11,
        "rankableBenchmarkCount": 11,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "frontierBench": 34.05,
          "terminalBench2": 84.3,
          "osWorldVerified": 85,
          "gdpvalAa": 1747,
          "aaAgenticIndex": 56.59,
          "tau2Bench": 98.5,
          "gdpvalAaNormalized": 61.1,
          "aaBriefcaseElo": 1572,
          "aaAutomationBench": 48.6,
          "aaEnterpriseOpsGym": 51.1,
          "aaHarveyLab": 93.6,
          "aaTau3Banking": 38.1,
          "terminalBenchHard": 62.9,
          "aaTerminalBench21": 84.6
        },
        "coding": {
          "sweVerified": 95,
          "swePro": 80,
          "frontierCode": 53.5,
          "terminalBench2": 84.3,
          "cursorBench31": 70.6,
          "cursorBench32": 70.5,
          "vulcanBench": 89.5,
          "aaCodingIndex": 76.49,
          "aaSciCode": 60.2
        },
        "reasoning": {
          "lcr": 76.7,
          "critpt": 28.6
        },
        "multimodalGrounded": {
          "blueprintBench2": 38.6,
          "officeQaPro": 57.9,
          "designArenaWebsite": 1315
        },
        "knowledge": {
          "artificialAnalysis": 62.07,
          "aaGpqaDiamond": 92.6,
          "aaHle": 55.5,
          "aaOmniscienceIndex": 43.3,
          "omniscienceAccuracy": 65.4,
          "omniscienceHallucinationRate": 63.6
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 63.5
        },
        "math": {}
      }
    },
    {
      "slug": "claude-opus-4-8",
      "canonicalModelKey": "claude-opus-4-8",
      "model": "Claude Opus 4.8",
      "creator": "Anthropic",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 75.96,
      "provisionalDisplayScore": 78,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 73.5,
        "upper": 78.43
      },
      "rankingEligible": true,
      "overallRank": 9,
      "url": "https://benchlm.ai/models/claude-opus-4-8",
      "markdownUrl": "https://benchlm.ai/md/models/claude-opus-4-8.md",
      "id": 238,
      "releaseDate": "2026-05-28",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "claude-opus-4-8",
        "familyName": "Claude Opus 4.8",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "claude-opus-4-8",
        "relatedModelKeys": [
          "claude-opus-4-7-max",
          "claude-opus-4-7",
          "claude-opus-4-6"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "claude-opus-4-7-max"
      },
      "scores": {
        "displayScore": 78,
        "overallScore": 78,
        "rawOverallScore": 76,
        "verifiedDisplayScore": 78,
        "displayCategoryScores": {
          "agentic": 83.1,
          "coding": 74.9,
          "reasoning": 68.3,
          "multimodalGrounded": 88.4,
          "knowledge": 86.5,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 66.8
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 83.1,
          "coding": 74.9,
          "reasoning": 68.3,
          "multimodalGrounded": 88.4,
          "knowledge": 86.5,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 66.8
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 9,
        "categoryRanks": {
          "agentic": 18,
          "coding": 8,
          "multimodalGrounded": 2,
          "knowledge": 4,
          "math": 2
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": true
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 30,
        "verifiedBenchmarkCount": 30,
        "rankableBenchmarkCount": 30,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 3
      },
      "benchmarks": {
        "agentic": {
          "frontierBench": 21.08,
          "terminalBench2": 74.6,
          "browseComp": 84.3,
          "deepSearchQa": 93.1,
          "osWorldVerified": 83.4,
          "financeAgentV2": 53.9,
          "gdpvalAa": 1593,
          "mcpAtlas": 82.2,
          "toolathlon": 59.9,
          "gertLabs": 72.97,
          "aaAgenticIndex": 49.37,
          "tau2Bench": 94.4,
          "gdpvalAaNormalized": 53.9,
          "researchClawBench": 21.1,
          "osWorld2": 20.6
        },
        "coding": {
          "sweVerified": 88.6,
          "swePro": 69.2,
          "sweMultilingual": 84.4,
          "sweMultimodal": 38.4,
          "terminalBench2": 74.6,
          "cursorBench31": 58.4,
          "cursorBench32": 62.3,
          "aaCodingIndex": 74.25,
          "aaSciCode": 53.5,
          "frontierCode": 46.5
        },
        "reasoning": {
          "arcAgi2": 72.08,
          "arcAgi3": 1.52,
          "lcr": 73,
          "critpt": 20.9
        },
        "multimodalGrounded": {
          "officeQaPro": 66.2,
          "screenSpotPro": 87.9,
          "charxiv": 89.9,
          "charxivNoTools": 80.5,
          "designArenaWebsite": 1271
        },
        "knowledge": {
          "gpqa": 93.6,
          "gpqaDiamond": 93.6,
          "hle": 57.9,
          "hleNoTools": 49.8,
          "artificialAnalysis": 57.33,
          "aaGpqaDiamond": 92,
          "aaHle": 48.7,
          "aaOmniscienceIndex": 28.8,
          "omniscienceAccuracy": 48.8,
          "omniscienceHallucinationRate": 39.3
        },
        "multilingual": {
          "include": 87.6
        },
        "instructionFollowing": {
          "aaIfBench": 62.2
        },
        "math": {
          "usamo2026": 96.7,
          "frontierMathV2Tiers13": 47.241,
          "frontierMathV2Tier4": 31.25
        }
      }
    },
    {
      "slug": "qwen3-8-max",
      "canonicalModelKey": "qwen3-8-max",
      "model": "Qwen3.8 Max",
      "creator": "Alibaba",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 78.55,
      "provisionalDisplayScore": 78,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 76.4,
        "upper": 80.7
      },
      "rankingEligible": true,
      "overallRank": 6,
      "url": "https://benchlm.ai/models/qwen3-8-max",
      "markdownUrl": "https://benchlm.ai/md/models/qwen3-8-max.md",
      "id": 298,
      "releaseDate": "2026-08-03",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "qwen3-8-max",
        "familyName": "Qwen3.8 Max",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "qwen3-8-max",
        "relatedModelKeys": [
          "qwen3-8-max-preview",
          "qwen3-7-max",
          "qwen3-6-max-preview"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "qwen3-8-max-preview"
      },
      "scores": {
        "displayScore": 78,
        "overallScore": 78,
        "rawOverallScore": 77,
        "verifiedDisplayScore": 78,
        "displayCategoryScores": {
          "agentic": 86.1,
          "coding": 59.5,
          "reasoning": 95.5,
          "multimodalGrounded": 87.1,
          "knowledge": 64.6,
          "multilingual": null,
          "instructionFollowing": 93.9,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 86.1,
          "coding": 59.2,
          "reasoning": 95.5,
          "multimodalGrounded": 87.1,
          "knowledge": 63.7,
          "multilingual": null,
          "instructionFollowing": 93.9,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 6,
        "categoryRanks": {
          "agentic": 4,
          "coding": 14,
          "reasoning": 1,
          "multimodalGrounded": 3,
          "knowledge": 33,
          "instructionFollowing": 2
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": true,
          "multimodalGrounded": true,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 42,
        "verifiedBenchmarkCount": 42,
        "rankableBenchmarkCount": 11,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "terminalBench21": 86.6,
          "coworkBench": 74.8,
          "jobBench": 53.4,
          "skillsBench": 70.2,
          "agentsLastExam": 52.4,
          "automationBench": 27.3,
          "toolathlonVerified": 72.5,
          "wideResearch": 81.9,
          "hleWithTools": 56.2,
          "osWorldVerified": 86.1,
          "osWorld2": 19.4,
          "webArenaVerified": 66.8,
          "androidWorld": 85.3,
          "mobileWorld": 77.8
        },
        "coding": {
          "terminalBench21": 86.6,
          "swePro": 67.7,
          "deepSwe": 56.6,
          "nl2Repo": 55.9,
          "frontierSwe": 73.5,
          "mlsBenchLite": 41,
          "paperBench": 93,
          "qwenReactBench": 1724,
          "vulcanBench": 81.2,
          "openHarmonyBench": 60.8
        },
        "reasoning": {
          "mrcrv2": 92.9,
          "longBenchV2": 66.3
        },
        "multimodalGrounded": {
          "mmmuPro": 82.3,
          "mathVision": 95.2,
          "mathVisionPython": 97.7,
          "babyVision": 82,
          "babyVisionPython": 91.3,
          "zeroBench": 24,
          "zeroBenchPython": 49,
          "medXpertQaMm": 80.4,
          "screenSpotPro": 84.5,
          "vision2Web": 69,
          "charxivNoTools": 88.4,
          "charxiv": 93.5,
          "omniDocBench15": 92.1,
          "ocrBenchV2": 74.2,
          "ccOcr": 79.6,
          "realWorldQa": 88,
          "erqa": 77.8,
          "simpleVqa": 75,
          "perceptionBench": 63.5,
          "videoMmeWithSub": 90.4,
          "videoMmmu": 88.7,
          "mmvu": 82.4,
          "mlvuAvg": 90.8,
          "lvBench": 81.8,
          "designArenaWebsite": 1299
        },
        "knowledge": {
          "gpqa": 92.6,
          "gpqaDiamond": 92.6,
          "hle": 43.6,
          "hleNoTools": 43.6
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifBench": 82.8
        },
        "math": {}
      }
    },
    {
      "slug": "qwen3-7-max",
      "canonicalModelKey": "qwen3-7-max",
      "model": "Qwen3.7 Max",
      "creator": "Alibaba",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 71.27,
      "provisionalDisplayScore": 76,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 63.59,
        "upper": 78.94
      },
      "rankingEligible": true,
      "overallRank": 18,
      "url": "https://benchlm.ai/models/qwen3-7-max",
      "markdownUrl": "https://benchlm.ai/md/models/qwen3-7-max.md",
      "id": 15,
      "releaseDate": "2026-05-16",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "qwen3-7-max",
        "familyName": "Qwen3.7 Max",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "qwen3-7-max",
        "relatedModelKeys": [
          "qwen3-6-plus",
          "qwen3-6-max-preview"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 76,
        "overallScore": 76,
        "rawOverallScore": 76,
        "verifiedDisplayScore": 76,
        "displayCategoryScores": {
          "agentic": 64,
          "coding": 81.7,
          "reasoning": 78,
          "multimodalGrounded": null,
          "knowledge": 70.1,
          "multilingual": 100,
          "instructionFollowing": 91,
          "math": 82.2
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 64,
          "coding": 81.7,
          "reasoning": 78,
          "multimodalGrounded": null,
          "knowledge": 70.1,
          "multilingual": 100,
          "instructionFollowing": 91,
          "math": 82.2
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 18,
        "categoryRanks": {
          "agentic": 119,
          "coding": 48,
          "knowledge": 23,
          "multilingual": 1,
          "instructionFollowing": 13
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": true,
          "multilingual": true,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 33,
        "verifiedBenchmarkCount": 33,
        "rankableBenchmarkCount": 33,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 4
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 69.7,
          "qwenClawBench": 64.3,
          "qwenWebBench": 1568,
          "clawEval": 65.2,
          "bfclV4": 75,
          "mcpAtlas": 76.4,
          "vitaBench": 47.9,
          "hleWithTools": 53.5,
          "aaAgenticIndex": 30.86,
          "tau2Bench": 94.7,
          "gdpvalAaNormalized": 38.2,
          "gdpvalAa": 1264,
          "gertLabs": 64.27,
          "researchClawBench": 18.7
        },
        "coding": {
          "sweVerified": 80.4,
          "swePro": 60.6,
          "sweMultilingual": 78.3,
          "nl2Repo": 47.2,
          "sciCode": 53.5,
          "liveCodeBench": 91.6,
          "terminalBench2": 69.7,
          "aaCodingIndex": 65.97,
          "aaSciCode": 48.8,
          "openHarmonyBench": 53.4
        },
        "reasoning": {
          "mrcrv2": 90.4,
          "critpt": 13.4,
          "lcr": 74.7
        },
        "multimodalGrounded": {
          "designArenaWebsite": 1291
        },
        "knowledge": {
          "gpqa": 92.4,
          "gpqaDiamond": 92.4,
          "hle": 41.4,
          "mmluPro": 89.6,
          "mmluRedux": 95,
          "superGpqa": 73.6,
          "mmmlu": 90.3,
          "artificialAnalysis": 46.71,
          "aaGpqaDiamond": 92.3,
          "aaHle": 40.5,
          "aaOmniscienceIndex": 13.5,
          "omniscienceAccuracy": 31.1,
          "omniscienceHallucinationRate": 25.6
        },
        "multilingual": {
          "mmluProX": 87,
          "nova63": 59,
          "include": 86.2,
          "maxife": 89.2,
          "polyMath": 86.5
        },
        "instructionFollowing": {
          "ifeval": 94.3,
          "ifBench": 79.1,
          "aaIfBench": 80.5
        },
        "math": {
          "hmmtFeb2026": 97.1,
          "imoAnswerBench": 90,
          "apex": 44.5
        }
      }
    },
    {
      "slug": "gpt-5-6-terra",
      "canonicalModelKey": "gpt-5-6-terra",
      "model": "GPT-5.6 Terra",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1.05M",
      "contextWindowTokens": 1050000,
      "displayScore": 72.5,
      "provisionalDisplayScore": 76,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 63.24,
        "upper": 81.76
      },
      "rankingEligible": true,
      "overallRank": 14,
      "url": "https://benchlm.ai/models/gpt-5-6-terra",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-5-6-terra.md",
      "id": 264,
      "releaseDate": "2026-07-09",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-5-6",
        "familyName": "GPT-5.6",
        "variantType": "terra",
        "snapshotLabel": "terra",
        "baseFamilyModelKey": "gpt-5-6-sol",
        "relatedModelKeys": [
          "gpt-5-6-sol",
          "gpt-5-6-luna"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 76,
        "overallScore": 76,
        "rawOverallScore": 75,
        "verifiedDisplayScore": 76,
        "displayCategoryScores": {
          "agentic": 91.2,
          "coding": 52.7,
          "reasoning": 78.1,
          "multimodalGrounded": 71.9,
          "knowledge": 84.2,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 97
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 91.2,
          "coding": 51.3,
          "reasoning": 78.1,
          "multimodalGrounded": 71.9,
          "knowledge": 84.2,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 97
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": false,
        "overallRank": 14,
        "categoryRanks": {
          "agentic": 26,
          "coding": 10,
          "multimodalGrounded": 11,
          "knowledge": 7
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 21,
        "verifiedBenchmarkCount": 21,
        "rankableBenchmarkCount": 21,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 3
      },
      "benchmarks": {
        "agentic": {
          "frontierBench": 20.81,
          "terminalBench2": 87.4,
          "browseComp": 87.5,
          "osWorld2": 50.2,
          "cyberGym": 81.8,
          "exploitGym": 23.2,
          "toolathlon": 53.1,
          "aaAgenticIndex": 50.2,
          "tau2Bench": 86.3,
          "gdpvalAaNormalized": 53.3,
          "gdpvalAa": 1583,
          "aaHarveyLab": 85.2,
          "aaItbench": 51,
          "aaTau3Banking": 40.2,
          "aaAutomationBench": 45.6,
          "terminalBenchHard": 57.6,
          "aaTerminalBench21": 88,
          "apexAgentsAa": 38.9,
          "aaBriefcaseElo": 1344,
          "aaEnterpriseOpsGym": 38.5
        },
        "coding": {
          "swePro": 63.4,
          "terminalBench2": 87.4,
          "deepSwe": 69.6,
          "frontierCode11Extended": 55.8,
          "cursorBench32": 64.9,
          "aaCodingIndex": 76.66,
          "aaSciCode": 53.9,
          "vulcanBench": 87
        },
        "reasoning": {
          "arcAgi2": 83.9,
          "arcAgi3": 0.8,
          "lcr": 79.7,
          "critpt": 30
        },
        "multimodalGrounded": {
          "mmmuPro": 80.7,
          "mmmuProPython": 82,
          "aaMmmuPro": 80.7
        },
        "knowledge": {
          "gpqa": 92.9,
          "gpqaDiamond": 92.9,
          "healthBenchProfessional": 57.7,
          "healthBenchHard": 32.7,
          "artificialAnalysis": 54.95,
          "aaGpqaDiamond": 92.5,
          "aaHle": 42.9,
          "aaOmniscienceIndex": 0.1,
          "omniscienceAccuracy": 46.8,
          "omniscienceHallucinationRate": 87.9
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 71.2
        },
        "math": {
          "frontierMath": 84.9,
          "frontierMathV2Tiers13": 84.9,
          "frontierMathV2Tier4": 68.3
        },
        "external": {
          "exploitBench": 52.9
        }
      }
    },
    {
      "slug": "claude-sonnet-5",
      "canonicalModelKey": "claude-sonnet-5",
      "model": "Claude Sonnet 5",
      "creator": "Anthropic",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 64.72,
      "provisionalDisplayScore": 75,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 51.18,
        "upper": 78.26
      },
      "rankingEligible": true,
      "overallRank": 39,
      "url": "https://benchlm.ai/models/claude-sonnet-5",
      "markdownUrl": "https://benchlm.ai/md/models/claude-sonnet-5.md",
      "id": 262,
      "releaseDate": "2026-06-30",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "claude-sonnet-5",
        "familyName": "Claude Sonnet 5",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "claude-sonnet-5",
        "relatedModelKeys": [
          "claude-sonnet-4-6",
          "claude-opus-4-8"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "claude-sonnet-4-6"
      },
      "scores": {
        "displayScore": 75,
        "overallScore": 75,
        "rawOverallScore": 73,
        "verifiedDisplayScore": 75,
        "displayCategoryScores": {
          "agentic": 86.2,
          "coding": 66.2,
          "reasoning": null,
          "multimodalGrounded": 76.5,
          "knowledge": 84.3,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 86.2,
          "coding": 66.2,
          "reasoning": null,
          "multimodalGrounded": 76.5,
          "knowledge": 84.3,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": false,
        "overallRank": 39,
        "categoryRanks": {
          "agentic": 11,
          "coding": 9,
          "knowledge": 6
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 15,
        "verifiedBenchmarkCount": 15,
        "rankableBenchmarkCount": 15,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "frontierBench": 14.59,
          "terminalBench2": 80.4,
          "browseComp": 84.7,
          "hleWithTools": 57.4,
          "osWorldVerified": 81.2,
          "gdpvalAa": 1603,
          "aaAgenticIndex": 49.72,
          "gdpvalAaNormalized": 54.2
        },
        "coding": {
          "sweVerified": 85.2,
          "swePro": 63.2,
          "sweMultilingual": 78.3,
          "sweMultimodal": 28.1,
          "terminalBench2": 80.4,
          "frontierCode": 42.7,
          "cursorBench32": 61.5,
          "aaCodingIndex": 71.55,
          "aaSciCode": 53.6
        },
        "reasoning": {
          "lcr": 77,
          "critpt": 16.9
        },
        "multimodalGrounded": {
          "charxiv": 88.3,
          "charxivNoTools": 77,
          "aaMmmuPro": 77.3,
          "designArenaWebsite": 1294
        },
        "knowledge": {
          "hle": 57.4,
          "hleNoTools": 43.2,
          "artificialAnalysis": 55.26,
          "aaGpqaDiamond": 91.1,
          "aaHle": 41.3,
          "aaOmniscienceIndex": 16.5,
          "omniscienceAccuracy": 40.1,
          "omniscienceHallucinationRate": 39.4
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "qwen3-7-plus",
      "canonicalModelKey": "qwen3-7-plus",
      "model": "Qwen3.7 Plus",
      "creator": "Alibaba",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 65.45,
      "provisionalDisplayScore": 72,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 54.69,
        "upper": 76.21
      },
      "rankingEligible": true,
      "overallRank": 36,
      "url": "https://benchlm.ai/models/qwen3-7-plus",
      "markdownUrl": "https://benchlm.ai/md/models/qwen3-7-plus.md",
      "id": 249,
      "releaseDate": "2026-06-03",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "qwen3-7-plus",
        "familyName": "Qwen3.7 Plus",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "qwen3-7-plus",
        "relatedModelKeys": [
          "qwen3-7-max",
          "qwen3-6-plus",
          "qwen3-6-max-preview"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "qwen3-6-plus"
      },
      "scores": {
        "displayScore": 72,
        "overallScore": 72,
        "rawOverallScore": 71,
        "verifiedDisplayScore": 72,
        "displayCategoryScores": {
          "agentic": 66.4,
          "coding": 70.1,
          "reasoning": 79.7,
          "multimodalGrounded": 67.7,
          "knowledge": 61.3,
          "multilingual": 78.9,
          "instructionFollowing": 91.4,
          "math": 78.4
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 66.4,
          "coding": 70.1,
          "reasoning": 79.7,
          "multimodalGrounded": 67.7,
          "knowledge": 61.3,
          "multilingual": 78.9,
          "instructionFollowing": 91.4,
          "math": 78.4
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 36,
        "categoryRanks": {
          "agentic": 115,
          "coding": 40,
          "multimodalGrounded": 14,
          "knowledge": 39,
          "multilingual": 3,
          "instructionFollowing": 8
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": true,
          "multilingual": true,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 50,
        "verifiedBenchmarkCount": 50,
        "rankableBenchmarkCount": 50,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 4
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 70.3,
          "qwenClawBench": 61.8,
          "qwenWebBench": 1536,
          "clawEval": 62.7,
          "bfclV4": 72.9,
          "mcpAtlas": 73.2,
          "vitaBench": 45.6,
          "deepPlanning": 62.3,
          "osWorldVerified": 73.3,
          "androidWorld": 81,
          "aaAgenticIndex": 20.69,
          "apexAgentsAa": 22.4,
          "tau2Bench": 93,
          "gdpvalAaNormalized": 22.4,
          "gdpvalAa": 947,
          "osWorld2": 2.8
        },
        "coding": {
          "terminalBench2": 70.3,
          "sweVerified": 77.7,
          "swePro": 57.6,
          "sweMultilingual": 75.8,
          "nl2Repo": 41.1,
          "sciCode": 51.3,
          "liveCodeBench": 89.6,
          "aaCodingIndex": 55.86,
          "aaSciCode": 45.5
        },
        "reasoning": {
          "critpt": 9.1,
          "mrcrv2": 91.7,
          "lcr": 69
        },
        "multimodalGrounded": {
          "mmmuPro": 79,
          "mathVision": 90.3,
          "charxiv": 85.9,
          "erqa": 69.8,
          "medXpertQaMm": 71,
          "screenSpotPro": 79,
          "simpleVqa": 81.7,
          "mmSearchPlus": 41.4,
          "realWorldQa": 86.9,
          "omniDocBench15": 91.4,
          "ocrBenchV2": 70.7,
          "odinw13": 51.1,
          "videoMmeWithSub": 88,
          "videoMmmu": 85.4,
          "mlvuAvg": 87.4,
          "aaMmmuPro": 80.5,
          "designArenaWebsite": 1288
        },
        "knowledge": {
          "gpqa": 90.3,
          "gpqaDiamond": 90.3,
          "hle": 34.7,
          "mmluPro": 88.5,
          "mmluRedux": 94.5,
          "superGpqa": 71.4,
          "mmmlu": 89,
          "artificialAnalysis": 39.37,
          "aaGpqaDiamond": 90,
          "aaHle": 35.6,
          "aaOmniscienceIndex": 1.1,
          "omniscienceAccuracy": 22.5,
          "omniscienceHallucinationRate": 27.7
        },
        "multilingual": {
          "mmluProX": 85.4,
          "nova63": 58.8,
          "include": 83,
          "maxife": 88.8,
          "polyMath": 84
        },
        "instructionFollowing": {
          "ifeval": 94.6,
          "ifBench": 79.1,
          "aaIfBench": 78
        },
        "math": {
          "hmmtFeb2026": 92.9,
          "imoAnswerBench": 86,
          "apex": 22.7
        }
      }
    },
    {
      "slug": "muse-spark-1-1",
      "canonicalModelKey": "muse-spark-1-1",
      "model": "Muse Spark 1.1",
      "creator": "Meta",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 76.52,
      "provisionalDisplayScore": 71,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 71.9,
        "upper": 81.14
      },
      "rankingEligible": true,
      "overallRank": 8,
      "url": "https://benchlm.ai/models/muse-spark-1-1",
      "markdownUrl": "https://benchlm.ai/md/models/muse-spark-1-1.md",
      "id": 281,
      "releaseDate": "2026-07-09",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "muse-spark",
        "familyName": "Muse Spark",
        "variantType": "1.1",
        "snapshotLabel": "1.1",
        "baseFamilyModelKey": "muse-spark-1-2",
        "relatedModelKeys": [
          "muse-spark-1-2",
          "muse-spark"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": "muse-spark"
      },
      "scores": {
        "displayScore": 71,
        "overallScore": 71,
        "rawOverallScore": 70,
        "verifiedDisplayScore": 71,
        "displayCategoryScores": {
          "agentic": 84,
          "coding": 49.8,
          "reasoning": null,
          "multimodalGrounded": 76.8,
          "knowledge": 92.5,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 84,
          "coding": 47.8,
          "reasoning": null,
          "multimodalGrounded": 76.8,
          "knowledge": 92.5,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": false,
        "overallRank": 8,
        "categoryRanks": {
          "agentic": 25,
          "coding": 16,
          "knowledge": 3
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 19,
        "verifiedBenchmarkCount": 19,
        "rankableBenchmarkCount": 19,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 80,
          "mcpAtlas": 88.1,
          "toolathlon": 75.6,
          "osWorldVerified": 80.8,
          "webArenaVerified": 69,
          "deepSearchQa": 84.9,
          "cyberGym": 59,
          "financeAgentV2": 57.2,
          "deepSwe": 53.3,
          "osWorld2": 14.2,
          "jobBench": 54.7,
          "cybench": 92.9,
          "exploitGym": 0.8,
          "aaAgenticIndex": 39.74,
          "gdpvalAaNormalized": 43.6,
          "gdpvalAa": 1375
        },
        "coding": {
          "terminalBench2": 80,
          "swePro": 61.5,
          "aaCodingIndex": 71.34,
          "aaSciCode": 58.2
        },
        "reasoning": {
          "mrcr1m": 54.1,
          "lcr": 81.3,
          "critpt": 15.1
        },
        "multimodalGrounded": {
          "charxiv": 88.4,
          "babyVision": 76.3,
          "designArenaWebsite": 1286
        },
        "knowledge": {
          "hle": 62.1,
          "hleNoTools": 52.2,
          "healthBenchProfessional": 59.3,
          "artificialAnalysis": 53.2,
          "aaGpqaDiamond": 89.8,
          "aaHle": 46.2,
          "aaOmniscienceIndex": 28.1,
          "omniscienceAccuracy": 52.1,
          "omniscienceHallucinationRate": 50
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "gemini-3-7-flash",
      "canonicalModelKey": "gemini-3-7-flash",
      "model": "Gemini 3.7 Flash",
      "creator": "Google",
      "sourceType": "Pending",
      "reasoningType": "Reasoning",
      "contextWindow": null,
      "contextWindowTokens": 0,
      "displayScore": 61,
      "provisionalDisplayScore": 71,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 49.48,
        "upper": 72.51
      },
      "rankingEligible": true,
      "overallRank": 63,
      "url": "https://benchlm.ai/models/gemini-3-7-flash",
      "markdownUrl": "https://benchlm.ai/md/models/gemini-3-7-flash.md",
      "id": 392,
      "releaseDate": "2026-08-13",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gemini-3-7-flash",
        "familyName": "Gemini 3.7 Flash",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gemini-3-7-flash",
        "relatedModelKeys": [
          "gemini-3-6-flash",
          "gemini-3-5-flash",
          "gemini-3-flash"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "gemini-3-6-flash"
      },
      "scores": {
        "displayScore": 71,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": 84.3,
          "coding": 58.5,
          "reasoning": null,
          "multimodalGrounded": 70.1,
          "knowledge": 66.5,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": false,
        "overallRank": 63,
        "categoryRanks": {
          "agentic": 42,
          "coding": 15,
          "multimodalGrounded": 12,
          "knowledge": 31
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "aaAgenticIndex": 45.1,
          "gdpvalAaNormalized": 50.8,
          "gdpvalAa": 1516,
          "aaBriefcaseElo": 1134,
          "aaAutomationBench": 62.7,
          "aaHarveyLab": 90.7,
          "aaTau3Banking": 32.8,
          "aaTerminalBench21": 85.8
        },
        "coding": {
          "aaCodingIndex": 76.12,
          "aaSciCode": 56.8
        },
        "reasoning": {
          "lcr": 80,
          "critpt": 14.3
        },
        "multimodalGrounded": {
          "aaMmmuPro": 85.5,
          "designArenaWebsite": 1321
        },
        "knowledge": {
          "artificialAnalysis": 56.03,
          "aaGpqaDiamond": 94.5,
          "aaHle": 47.9,
          "aaOmniscienceIndex": 26.5,
          "omniscienceAccuracy": 55.3,
          "omniscienceHallucinationRate": 64.5
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "gpt-5-5",
      "canonicalModelKey": "gpt-5-5",
      "model": "GPT-5.5",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 72.68,
      "provisionalDisplayScore": 70,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 64.14,
        "upper": 81.23
      },
      "rankingEligible": true,
      "overallRank": 13,
      "url": "https://benchlm.ai/models/gpt-5-5",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-5-5.md",
      "id": 32,
      "releaseDate": "2026-04-23",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-5-5",
        "familyName": "GPT-5.5",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-5-5",
        "relatedModelKeys": [
          "gpt-5-5-pro"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "gpt-5-4"
      },
      "scores": {
        "displayScore": 70,
        "overallScore": 69,
        "rawOverallScore": 69,
        "verifiedDisplayScore": 69,
        "displayCategoryScores": {
          "agentic": 85,
          "coding": 44.5,
          "reasoning": 79,
          "multimodalGrounded": 65.8,
          "knowledge": 77.5,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 71.1
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 85,
          "coding": 42.5,
          "reasoning": 79,
          "multimodalGrounded": 65.8,
          "knowledge": 77.5,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 71.1
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 13,
        "categoryRanks": {
          "agentic": 14,
          "coding": 7,
          "multimodalGrounded": 15,
          "knowledge": 18
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 29,
        "verifiedBenchmarkCount": 29,
        "rankableBenchmarkCount": 29,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 3
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 82,
          "cyberGym": 81.8,
          "browseComp": 84.4,
          "osWorldVerified": 78.7,
          "mcpAtlas": 75.3,
          "toolathlon": 55.6,
          "tau2Bench": 98,
          "aaAgenticIndex": 47.41,
          "apexAgentsAa": 37.7,
          "gdpvalAaNormalized": 49.1,
          "gdpvalAa": 1482,
          "gertLabs": 72.93,
          "researchClawBench": 17,
          "osWorld2": 13,
          "jobBench": 42.7,
          "exploitGym": 13.4,
          "aaItbench": 45.8
        },
        "coding": {
          "swePro": 58.6,
          "terminalBench2": 82,
          "vibeCodeBench": 69.847,
          "reactNativeEvals": 84.7,
          "cursorBench31": 59.2,
          "cursorBench32": 58.4,
          "aaCodingIndex": 74.89,
          "aaSciCode": 56.1,
          "frontierCode": 43
        },
        "reasoning": {
          "mrcrv2_64_128": 83.1,
          "mrcrv2_128_256": 87.5,
          "arcAgi2": 85,
          "arcAgi3": 0.43,
          "lcr": 79,
          "critpt": 27.1
        },
        "multimodalGrounded": {
          "mmmuPro": 81.2,
          "mmmuProPython": 83.2,
          "officeQaPro": 54.1,
          "aaMmmuPro": 79.9,
          "designArenaWebsite": 1274
        },
        "knowledge": {
          "gpqa": 93.6,
          "gpqaDiamond": 93.6,
          "hle": 52.2,
          "hleNoTools": 41.4,
          "artificialAnalysis": 56.31,
          "aaGpqaDiamond": 93.5,
          "aaHle": 45.8,
          "aaOmniscienceIndex": 20.5,
          "omniscienceAccuracy": 58,
          "omniscienceHallucinationRate": 89
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 75.9
        },
        "math": {
          "frontierMath": 51.7,
          "frontierMathV2Tiers13": 51.7,
          "frontierMathV2Tier4": 35.4
        },
        "external": {
          "secBenchPro": 45.8,
          "atomicNetworkAttackSimulation": 100,
          "atomicVulnerabilityResearch": 92,
          "atomicEvasion": 54
        }
      }
    },
    {
      "slug": "gpt-5-6-luna",
      "canonicalModelKey": "gpt-5-6-luna",
      "model": "GPT-5.6 Luna",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1.05M",
      "contextWindowTokens": 1050000,
      "displayScore": 66.93,
      "provisionalDisplayScore": 69,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 57.29,
        "upper": 76.58
      },
      "rankingEligible": true,
      "overallRank": 28,
      "url": "https://benchlm.ai/models/gpt-5-6-luna",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-5-6-luna.md",
      "id": 265,
      "releaseDate": "2026-07-09",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-5-6",
        "familyName": "GPT-5.6",
        "variantType": "luna",
        "snapshotLabel": "luna",
        "baseFamilyModelKey": "gpt-5-6-sol",
        "relatedModelKeys": [
          "gpt-5-6-sol",
          "gpt-5-6-terra"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 69,
        "overallScore": 69,
        "rawOverallScore": 68,
        "verifiedDisplayScore": 69,
        "displayCategoryScores": {
          "agentic": 84.7,
          "coding": 52,
          "reasoning": 57.9,
          "multimodalGrounded": 60.7,
          "knowledge": 83.6,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 97
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 84.7,
          "coding": 50,
          "reasoning": 57.9,
          "multimodalGrounded": 60.7,
          "knowledge": 83.6,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 97
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": false,
        "overallRank": 28,
        "categoryRanks": {
          "agentic": 58,
          "coding": 6,
          "multimodalGrounded": 21,
          "knowledge": 9
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 21,
        "verifiedBenchmarkCount": 21,
        "rankableBenchmarkCount": 21,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 3
      },
      "benchmarks": {
        "agentic": {
          "frontierBench": 14.32,
          "terminalBench2": 84.7,
          "browseComp": 83.3,
          "osWorld2": 45.6,
          "cyberGym": 77.9,
          "exploitGym": 12.4,
          "toolathlon": 53.4,
          "aaAgenticIndex": 46.9,
          "gdpvalAaNormalized": 53.5,
          "gdpvalAa": 1582,
          "aaHarveyLab": 87.9,
          "aaItbench": 40.3,
          "aaTau3Banking": 31.1,
          "aaAutomationBench": 42.2,
          "aaTerminalBench21": 80.9,
          "apexAgentsAa": 35.8,
          "aaBriefcaseElo": 1354,
          "aaEnterpriseOpsGym": 40.8
        },
        "coding": {
          "swePro": 62.7,
          "terminalBench2": 84.7,
          "deepSwe": 67.2,
          "frontierCode11Extended": 55.1,
          "cursorBench32": 61.1,
          "aaCodingIndex": 71.45,
          "aaSciCode": 52.5,
          "vulcanBench": 85.5
        },
        "reasoning": {
          "arcAgi2": 59.54,
          "arcAgi3": 0.18,
          "lcr": 78.3,
          "critpt": 20.6
        },
        "multimodalGrounded": {
          "mmmuPro": 78.4,
          "mmmuProPython": 79.5,
          "aaMmmuPro": 78.6
        },
        "knowledge": {
          "gpqa": 92.3,
          "gpqaDiamond": 92.3,
          "healthBenchProfessional": 55.7,
          "healthBenchHard": 32,
          "artificialAnalysis": 51.24,
          "aaGpqaDiamond": 91.1,
          "aaHle": 39.5,
          "aaOmniscienceIndex": -10.3,
          "omniscienceAccuracy": 42.7,
          "omniscienceHallucinationRate": 92.6
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {
          "frontierMath": 78.6,
          "frontierMathV2Tiers13": 78.6,
          "frontierMathV2Tier4": 58.5
        },
        "external": {
          "exploitBench": 33.2
        }
      }
    },
    {
      "slug": "glm-5-2",
      "canonicalModelKey": "glm-5-2",
      "model": "GLM-5.2",
      "creator": "Z.AI",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 62.88,
      "provisionalDisplayScore": 68,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 48.66,
        "upper": 77.1
      },
      "rankingEligible": true,
      "overallRank": 48,
      "url": "https://benchlm.ai/models/glm-5-2",
      "markdownUrl": "https://benchlm.ai/md/models/glm-5-2.md",
      "id": 259,
      "releaseDate": "2026-06-16",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "glm-5",
        "familyName": "GLM-5",
        "variantType": "flagship",
        "snapshotLabel": "5.2",
        "baseFamilyModelKey": "glm-5",
        "relatedModelKeys": [
          "glm-5-1",
          "glm-5",
          "glm-5-reasoning",
          "glm-5-turbo",
          "glm-5-3"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "glm-5-1"
      },
      "scores": {
        "displayScore": 68,
        "overallScore": 68,
        "rawOverallScore": 68,
        "verifiedDisplayScore": 68,
        "displayCategoryScores": {
          "agentic": 80.8,
          "coding": 50.9,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 81,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 80.7
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 80.8,
          "coding": 48.9,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 81,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 80.7
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 48,
        "categoryRanks": {
          "agentic": 22,
          "coding": 21,
          "knowledge": 13
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 15,
        "verifiedBenchmarkCount": 15,
        "rankableBenchmarkCount": 15,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "frontierBench": 4.59,
          "terminalBench2": 81,
          "mcpAtlas": 76.8,
          "toolathlon": 48.2,
          "aaAgenticIndex": 45.67,
          "tau2Bench": 99.1,
          "gdpvalAaNormalized": 49.9,
          "gdpvalAa": 1498,
          "apexAgentsAa": 33.7,
          "researchClawBench": 20.7
        },
        "coding": {
          "swePro": 62.1,
          "nl2Repo": 48.9,
          "terminalBench2": 81,
          "programBench": 63.7,
          "cursorBench32": 55,
          "aaCodingIndex": 68.76,
          "aaSciCode": 50.5,
          "openHarmonyBench": 58.4
        },
        "reasoning": {
          "critpt": 20.9,
          "lcr": 76.7
        },
        "multimodalGrounded": {
          "designArenaWebsite": 1320
        },
        "knowledge": {
          "gpqa": 91.2,
          "gpqaDiamond": 91.2,
          "hle": 54.7,
          "hleNoTools": 40.5,
          "artificialAnalysis": 52.64,
          "aaGpqaDiamond": 89.5,
          "aaHle": 41.1,
          "aaOmniscienceIndex": 4.4,
          "omniscienceAccuracy": 24.3,
          "omniscienceHallucinationRate": 26.3
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 73.3
        },
        "math": {
          "aime2026": 99.2,
          "hmmtNov2025": 94.4,
          "hmmtFeb2026": 92.5,
          "mmAnswerBench": 91
        }
      }
    },
    {
      "slug": "claude-opus-4-7-adaptive",
      "canonicalModelKey": "claude-opus-4-7-max",
      "model": "Claude Opus 4.7 (Adaptive)",
      "creator": "Anthropic",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 72.21,
      "provisionalDisplayScore": 68,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 62.34,
        "upper": 82.07
      },
      "rankingEligible": true,
      "overallRank": 16,
      "url": "https://benchlm.ai/models/claude-opus-4-7-adaptive",
      "markdownUrl": "https://benchlm.ai/md/models/claude-opus-4-7-adaptive.md",
      "id": 33,
      "releaseDate": "2026-04-16",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "claude-opus-4-7",
        "familyName": "Claude Opus 4.7",
        "variantType": "reasoning",
        "snapshotLabel": "adaptive",
        "baseFamilyModelKey": "claude-opus-4-7",
        "relatedModelKeys": [
          "claude-opus-4-7",
          "claude-opus-4-6",
          "claude-opus-4-5"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": "claude-opus-4-6"
      },
      "scores": {
        "displayScore": 68,
        "overallScore": 68,
        "rawOverallScore": 67,
        "verifiedDisplayScore": 68,
        "displayCategoryScores": {
          "agentic": 70.4,
          "coding": 69.8,
          "reasoning": 71.4,
          "multimodalGrounded": 49.1,
          "knowledge": 81.6,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 70.4,
          "coding": 69.8,
          "reasoning": 71.4,
          "multimodalGrounded": 49.1,
          "knowledge": 81.6,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 16,
        "categoryRanks": {
          "agentic": 15,
          "coding": 19,
          "multimodalGrounded": 24,
          "knowledge": 12
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 20,
        "verifiedBenchmarkCount": 20,
        "rankableBenchmarkCount": 20,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 3
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 69.4,
          "browseComp": 79.3,
          "mcpAtlas": 77.3,
          "osWorldVerified": 78,
          "cyberGym": 73.1,
          "aaAgenticIndex": 46.31,
          "tau2Bench": 88.6,
          "gdpvalAaNormalized": 49.1,
          "gdpvalAa": 1483,
          "osWorld2": 18.2,
          "jobBench": 45.9,
          "aaItbench": 46.7
        },
        "coding": {
          "sweVerified": 87.6,
          "swePro": 64.3,
          "terminalBench2": 69.4,
          "aaCodingIndex": 73.6,
          "aaSciCode": 54.5
        },
        "reasoning": {
          "mrcrv2_128_256": 59.2,
          "arcAgi2": 75.8,
          "arcAgi3": 0.18,
          "lcr": 75.3,
          "critpt": 12
        },
        "multimodalGrounded": {
          "officeQaPro": 43.6,
          "charxiv": 91,
          "charxivNoTools": 82.1,
          "aaMmmuPro": 78.8,
          "designArenaWebsite": 1313
        },
        "knowledge": {
          "gpqa": 94.2,
          "gpqaDiamond": 94.2,
          "hle": 54.7,
          "hleNoTools": 46.9,
          "artificialAnalysis": 54.96,
          "aaGpqaDiamond": 91.4,
          "aaHle": 42.3,
          "aaOmniscienceIndex": 27.3,
          "omniscienceAccuracy": 48.9,
          "omniscienceHallucinationRate": 42.3
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 58.6
        },
        "math": {
          "frontierMath": 43.8
        }
      }
    },
    {
      "slug": "gemini-3-1-pro",
      "canonicalModelKey": "gemini-3-1-pro",
      "model": "Gemini 3.1 Pro",
      "creator": "Google",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 56.21,
      "provisionalDisplayScore": 68,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 40.17,
        "upper": 72.24
      },
      "rankingEligible": true,
      "overallRank": 98,
      "url": "https://benchlm.ai/models/gemini-3-1-pro",
      "markdownUrl": "https://benchlm.ai/md/models/gemini-3-1-pro.md",
      "id": 14,
      "releaseDate": "2026-02-19",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gemini-3-1-pro",
        "familyName": "Gemini 3.1 Pro",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gemini-3-1-pro",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 68,
        "overallScore": 68,
        "rawOverallScore": 67,
        "verifiedDisplayScore": 68,
        "displayCategoryScores": {
          "agentic": 67.8,
          "coding": 59.2,
          "reasoning": 72.4,
          "multimodalGrounded": 78.2,
          "knowledge": 66.4,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 55.1
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": 72.4,
          "multimodalGrounded": 78.2,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 55.1
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": false,
        "overallRank": 98,
        "categoryRanks": {
          "agentic": 128,
          "coding": 85,
          "multimodalGrounded": 9,
          "knowledge": 32
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 20,
        "verifiedBenchmarkCount": 20,
        "rankableBenchmarkCount": 20,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 3
      },
      "benchmarks": {
        "agentic": {
          "clawEval": 57.8,
          "deepSearchQa": 69.7,
          "tau2Bench": 95.6,
          "aaAgenticIndex": 23.05,
          "apexAgentsAa": 32,
          "gdpvalAaNormalized": 23.2,
          "gdpvalAa": 965,
          "gertLabs": 56.87,
          "researchClawBench": 13.3
        },
        "coding": {
          "liveCodeBenchPro": 82.9,
          "reactNativeEvals": 78.9,
          "vibeCodeBench": 32.034,
          "aaCodingIndex": 68.83,
          "aaSciCode": 58.9
        },
        "reasoning": {
          "arcAgi2": 77.08,
          "arcAgi3": 0.42,
          "lcr": 79,
          "critpt": 17.7
        },
        "multimodalGrounded": {
          "mmmuPro": 83.9,
          "charxiv": 80.2,
          "erqa": 69.4,
          "simpleVqa": 72.4,
          "screenSpotPro": 84.4,
          "zeroBench": 29,
          "medXpertQaMm": 81.3,
          "aaMmmuPro": 82.4,
          "designArenaWebsite": 1271
        },
        "knowledge": {
          "gpqaDiamond": 94.3,
          "hleNoTools": 45.4,
          "healthBenchHard": 20.6,
          "medXpertQaText": 71.5,
          "artificialAnalysis": 47.74,
          "aaGpqaDiamond": 94.1,
          "aaHle": 47,
          "aaOmniscienceIndex": 31.9,
          "omniscienceAccuracy": 54.9,
          "omniscienceHallucinationRate": 50.9
        },
        "multilingual": {
          "aaGlobalMmluLite": 93.2
        },
        "instructionFollowing": {
          "aaIfBench": 77.1
        },
        "math": {
          "frontierMathV2Tiers13": 36.9,
          "frontierMathV2Tier4": 16.7
        }
      }
    },
    {
      "slug": "muse-spark-1-2",
      "canonicalModelKey": "muse-spark-1-2",
      "model": "Muse Spark 1.2",
      "creator": "Meta",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 61.29,
      "provisionalDisplayScore": 67,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 49.78,
        "upper": 72.81
      },
      "rankingEligible": true,
      "overallRank": 55,
      "url": "https://benchlm.ai/models/muse-spark-1-2",
      "markdownUrl": "https://benchlm.ai/md/models/muse-spark-1-2.md",
      "id": 385,
      "releaseDate": "2026-08-05",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "muse-spark",
        "familyName": "Muse Spark",
        "variantType": "1.2",
        "snapshotLabel": "1.2",
        "baseFamilyModelKey": "muse-spark-1-2",
        "relatedModelKeys": [
          "muse-spark-1-1",
          "muse-spark"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "muse-spark-1-1"
      },
      "scores": {
        "displayScore": 67,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": 76,
          "coding": 59.2,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 60.6,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": false,
        "overallRank": 55,
        "categoryRanks": {
          "agentic": 39,
          "coding": 27,
          "knowledge": 41
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "terminalBench21": 82.9,
          "gdpvalAa": 1631,
          "aaAgenticIndex": 49.31,
          "gdpvalAaNormalized": 55.8,
          "aaBriefcaseElo": 1361,
          "aaTau3Banking": 34.8,
          "aaTerminalBench21": 80.1,
          "aaEnterpriseOpsGym": 47.3
        },
        "coding": {
          "terminalBench21": 82.9,
          "deepSwe": 59.3,
          "aaCodingIndex": 72.22,
          "aaSciCode": 56.4
        },
        "reasoning": {
          "lcr": 83.3,
          "critpt": 17.7
        },
        "multimodalGrounded": {
          "designArenaWebsite": 1332
        },
        "knowledge": {
          "artificialAnalysis": 56.76,
          "aaGpqaDiamond": 90.4,
          "aaHle": 45.5,
          "aaOmniscienceIndex": 27.2,
          "omniscienceAccuracy": 45.4,
          "omniscienceHallucinationRate": 33.3
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "gemini-3-5-flash",
      "canonicalModelKey": "gemini-3-5-flash",
      "model": "Gemini 3.5 Flash",
      "creator": "Google",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 64.2,
      "provisionalDisplayScore": 67,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 54.19,
        "upper": 74.21
      },
      "rankingEligible": true,
      "overallRank": 43,
      "url": "https://benchlm.ai/models/gemini-3-5-flash",
      "markdownUrl": "https://benchlm.ai/md/models/gemini-3-5-flash.md",
      "id": 38,
      "releaseDate": "2026-05-19",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gemini-3-5-flash",
        "familyName": "Gemini 3.5 Flash",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gemini-3-5-flash",
        "relatedModelKeys": [
          "gemini-3-flash",
          "gemini-3-1-pro"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "gemini-3-flash"
      },
      "scores": {
        "displayScore": 67,
        "overallScore": 66,
        "rawOverallScore": 66,
        "verifiedDisplayScore": 66,
        "displayCategoryScores": {
          "agentic": 77.7,
          "coding": 55,
          "reasoning": 62.6,
          "multimodalGrounded": 82.4,
          "knowledge": 60.2,
          "multilingual": null,
          "instructionFollowing": 82.4,
          "math": 55.9
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 77.7,
          "coding": 54,
          "reasoning": 62.6,
          "multimodalGrounded": 82.4,
          "knowledge": 58.2,
          "multilingual": null,
          "instructionFollowing": 82.4,
          "math": 55.9
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 43,
        "categoryRanks": {
          "agentic": 89,
          "coding": 29,
          "reasoning": 2,
          "multimodalGrounded": 6,
          "knowledge": 42,
          "instructionFollowing": 24
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": true,
          "multimodalGrounded": true,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 23,
        "verifiedBenchmarkCount": 23,
        "rankableBenchmarkCount": 23,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 4
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 76.2,
          "mcpAtlas": 83.6,
          "toolathlon": 56.5,
          "osWorldVerified": 78.4,
          "financeAgentV2": 57.861,
          "gdpvalAa": 1345,
          "tau2Bench": 95.3,
          "gdpvalAaNormalized": 42.2,
          "aaAgenticIndex": 39.71,
          "apexAgentsAa": 47.1,
          "gertLabs": 61.85,
          "researchClawBench": 18,
          "aaEnterpriseOpsGym": 50.1
        },
        "coding": {
          "terminalBench2": 76.2,
          "swePro": 55.1,
          "sciCode": 53.1,
          "vibeCodeBench": 48.683,
          "cursorBench31": 49.8,
          "cursorBench32": 48.8,
          "aaCodingIndex": 70.14,
          "aaSciCode": 53.1
        },
        "reasoning": {
          "mrcrv2": 77.3,
          "mrcr1m": 26.6,
          "arcAgi2": 72.1,
          "lcr": 69.3,
          "critpt": 13.1
        },
        "multimodalGrounded": {
          "charxiv": 84.2,
          "mmmuPro": 83.6,
          "blueprintBench2": 33.6,
          "aaMmmuPro": 84.3,
          "designArenaWebsite": 1280
        },
        "knowledge": {
          "artificialAnalysis": 50.2,
          "gpqa": 92.2,
          "gpqaDiamond": 92.676,
          "hle": 40.2,
          "omniscienceAccuracy": 51.9,
          "omniscienceHallucinationRate": 60.7,
          "aaGpqaDiamond": 92.2,
          "aaHle": 42.7,
          "aaOmniscienceIndex": 21.2
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifBench": 76.3,
          "aaIfBench": 76.3
        },
        "math": {
          "frontierMathV2Tiers13": 38.966,
          "frontierMathV2Tier4": 14.583
        }
      }
    },
    {
      "slug": "claude-opus-4-6",
      "canonicalModelKey": "claude-opus-4-6",
      "model": "Claude Opus 4.6",
      "creator": "Anthropic",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 67.89,
      "provisionalDisplayScore": 66,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 53.87,
        "upper": 81.9
      },
      "rankingEligible": true,
      "overallRank": 23,
      "url": "https://benchlm.ai/models/claude-opus-4-6",
      "markdownUrl": "https://benchlm.ai/md/models/claude-opus-4-6.md",
      "id": 18,
      "releaseDate": "2026-02-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "claude-opus-4-6",
        "familyName": "Claude Opus 4.6",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "claude-opus-4-6",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 66,
        "overallScore": 66,
        "rawOverallScore": 65,
        "verifiedDisplayScore": 66,
        "displayCategoryScores": {
          "agentic": 65.5,
          "coding": 66.3,
          "reasoning": null,
          "multimodalGrounded": 55.3,
          "knowledge": 82.2,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 59.6
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 65.5,
          "coding": 66.3,
          "reasoning": null,
          "multimodalGrounded": 55.3,
          "knowledge": 82.2,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 59.6
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 23,
        "categoryRanks": {
          "agentic": 37,
          "coding": 31,
          "knowledge": 11
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 30,
        "verifiedBenchmarkCount": 30,
        "rankableBenchmarkCount": 30,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 3
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 65.4,
          "browseComp": 83.7,
          "osWorldVerified": 72.7,
          "tau2Bench": 84.8,
          "clawEval": 70.4,
          "deepSearchQa": 73.7,
          "cyberGym": 66.6,
          "gertLabs": 61.85,
          "researchClawBench": 19.9,
          "jobBench": 36.7
        },
        "coding": {
          "sweVerified": 80.84,
          "sweVerifiedArcee": 75.6,
          "liveCodeBenchPro": 70.7,
          "swePro": 53.4,
          "sweRebench": 65.3,
          "reactNativeEvals": 84.1,
          "vibeCodeBench": 57.573,
          "aaSciCode": 45.7,
          "frontierCode": 26.9
        },
        "reasoning": {
          "lcr": 62.3,
          "critpt": 2.8
        },
        "multimodalGrounded": {
          "mmmuPro": 77.3,
          "erqa": 51.6,
          "screenSpotPro": 83.1,
          "medXpertQaMm": 64.8,
          "aaMmmuPro": 72.5,
          "designArenaWebsite": 1309
        },
        "knowledge": {
          "gpqa": 91.3,
          "gpqaDiamond": 89.2,
          "superGpqa": 95,
          "mmluPro": 82,
          "mmluProArcee": 89.1,
          "hle": 53,
          "hleNoTools": 40,
          "healthBenchHard": 14.8,
          "medXpertQaText": 52.1,
          "artificialAnalysis": 38.77,
          "aaGpqaDiamond": 84,
          "aaHle": 19.1,
          "aaOmniscienceIndex": 2.4,
          "omniscienceAccuracy": 45.8,
          "omniscienceHallucinationRate": 80.1
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 44.6
        },
        "math": {
          "aime2025Arcee": 99.8,
          "frontierMathV2Tiers13": 40.7,
          "frontierMathV2Tier4": 22.9
        }
      }
    },
    {
      "slug": "deepseek-v4-pro-0813",
      "canonicalModelKey": "deepseek-v4-pro-max",
      "model": "DeepSeek V4 Pro 0813",
      "creator": "DeepSeek",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 61.22,
      "provisionalDisplayScore": 64,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 51.35,
        "upper": 71.09
      },
      "rankingEligible": true,
      "overallRank": 57,
      "url": "https://benchlm.ai/models/deepseek-v4-pro-0813",
      "markdownUrl": "https://benchlm.ai/md/models/deepseek-v4-pro-0813.md",
      "id": 29,
      "releaseDate": "2026-08-13",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "deepseek-v4-pro",
        "familyName": "DeepSeek V4 Pro",
        "variantType": "pro-reasoning",
        "snapshotLabel": "0813",
        "baseFamilyModelKey": "deepseek-v4-pro-max",
        "relatedModelKeys": [
          "deepseek-v4-pro-base",
          "deepseek-v4-pro",
          "deepseek-v4-pro-high"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "deepseek-v4-pro"
      },
      "scores": {
        "displayScore": 64,
        "overallScore": 64,
        "rawOverallScore": 63,
        "verifiedDisplayScore": 64,
        "displayCategoryScores": {
          "agentic": 67.5,
          "coding": 55.6,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 70,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 80.5
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 67.5,
          "coding": 54.7,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 70,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 80.5
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 57,
        "categoryRanks": {
          "agentic": 30,
          "coding": 78,
          "knowledge": 24
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 23,
        "verifiedBenchmarkCount": 23,
        "rankableBenchmarkCount": 23,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 67.9,
          "terminalBench21": 87.9,
          "browseComp": 83.4,
          "hleWithTools": 60,
          "mcpAtlas": 73.6,
          "gdpvalAa": 1306,
          "toolathlon": 51.8,
          "aaAgenticIndex": 49.56,
          "apexAgentsAa": 24.3,
          "tau2Bench": 96.2,
          "gdpvalAaNormalized": 54.5,
          "cyberGym": 83.3,
          "toolathlonVerified": 74.1,
          "agentsLastExam": 25.7,
          "automationBench": 31.8,
          "aaTau3Banking": 39.6,
          "aaTerminalBench21": 78.7,
          "aaBriefcaseElo": 1286,
          "aaEnterpriseOpsGym": 49.6
        },
        "coding": {
          "liveCodeBenchPass1Cot": 93.5,
          "codeforces": 3206,
          "sweVerified": 80.6,
          "swePro": 55.4,
          "sweMultilingual": 76.2,
          "terminalBench2": 67.9,
          "vibeCodeBench": 49.931,
          "aaCodingIndex": 68.83,
          "aaSciCode": 49.2,
          "terminalBench21": 87.9,
          "nl2Repo": 61.5,
          "deepSwe": 62.7,
          "dsBenchFullStack": 71.1,
          "dsBenchHard": 67.2
        },
        "reasoning": {
          "mrcr1m": 83.5,
          "corpusQa1m": 62,
          "lcr": 75.3,
          "critpt": 18
        },
        "multimodalGrounded": {
          "designArenaWebsite": 1258
        },
        "knowledge": {
          "mmluPro": 87.5,
          "simpleQa": 57.9,
          "chineseSimpleQa": 84.4,
          "gpqa": 90.1,
          "gpqaDiamond": 90.1,
          "hle": 42.7,
          "artificialAnalysis": 53.2,
          "aaGpqaDiamond": 92.8,
          "aaHle": 41,
          "aaOmniscienceIndex": 0.8,
          "omniscienceAccuracy": 49.1,
          "omniscienceHallucinationRate": 94.1,
          "aaOpennessIndex": 44.4
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 76.5
        },
        "math": {
          "hmmtFeb2026": 95.2,
          "imoAnswerBench": 89.8,
          "apex": 38.3,
          "apexShortlist": 90.2
        }
      }
    },
    {
      "slug": "gpt-5-4",
      "canonicalModelKey": "gpt-5-4",
      "model": "GPT-5.4",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1.05M",
      "contextWindowTokens": 1050000,
      "displayScore": 73.04,
      "provisionalDisplayScore": 64,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 69.63,
        "upper": 76.45
      },
      "rankingEligible": true,
      "overallRank": 12,
      "url": "https://benchlm.ai/models/gpt-5-4",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-5-4.md",
      "id": 11,
      "releaseDate": "2026-03-05",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-5-4",
        "familyName": "GPT-5.4",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-5-4",
        "relatedModelKeys": [
          "gpt-5-4-pro",
          "gpt-5-4-mini",
          "gpt-5-4-nano"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 64,
        "overallScore": 64,
        "rawOverallScore": 64,
        "verifiedDisplayScore": 64,
        "displayCategoryScores": {
          "agentic": 74.8,
          "coding": 42.9,
          "reasoning": 69.8,
          "multimodalGrounded": 64.8,
          "knowledge": 75.1,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 65.7
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 74.8,
          "coding": 40.9,
          "reasoning": 69.8,
          "multimodalGrounded": 64.8,
          "knowledge": 75.1,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 65.7
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 12,
        "categoryRanks": {
          "agentic": 31,
          "coding": 38,
          "multimodalGrounded": 16,
          "knowledge": 21
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 32,
        "verifiedBenchmarkCount": 32,
        "rankableBenchmarkCount": 32,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 3
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 75.1,
          "cyberGym": 79,
          "browseComp": 82.7,
          "osWorldVerified": 75,
          "mcpAtlas": 70.6,
          "toolathlon": 54.6,
          "tau2Bench": 98.9,
          "clawEval": 60.3,
          "deepSearchQa": 73.6,
          "aaAgenticIndex": 44.17,
          "apexAgentsAa": 33.3,
          "gdpvalAaNormalized": 44.2,
          "gdpvalAa": 1385,
          "gertLabs": 64.89,
          "researchClawBench": 15.3,
          "jobBench": 38.9,
          "exploitGym": 6
        },
        "coding": {
          "liveCodeBenchPro": 87.5,
          "swePro": 57.7,
          "reactNativeEvals": 85.3,
          "vibeCodeBench": 67.421,
          "aaCodingIndex": 71.05,
          "aaSciCode": 56.6
        },
        "reasoning": {
          "arcAgi2": 73.95,
          "arcAgi3": 0.21,
          "lcr": 77.7,
          "critpt": 23.4
        },
        "multimodalGrounded": {
          "mmmuPro": 81.2,
          "officeQaPro": 53.2,
          "mmmuProPython": 82.1,
          "charxiv": 82.8,
          "erqa": 65.4,
          "simpleVqa": 61.1,
          "screenSpotPro": 85.4,
          "zeroBench": 41,
          "medXpertQaMm": 77.1,
          "aaMmmuPro": 78.4,
          "designArenaWebsite": 1238
        },
        "knowledge": {
          "gpqa": 92.8,
          "hle": 52.1,
          "hleNoTools": 39.8,
          "gpqaDiamond": 92.8,
          "healthBenchHard": 40.1,
          "medXpertQaText": 59.6,
          "artificialAnalysis": 53.12,
          "aaGpqaDiamond": 92,
          "aaHle": 43.7,
          "aaOmniscienceIndex": 5.8,
          "omniscienceAccuracy": 50.8,
          "omniscienceHallucinationRate": 91.7,
          "healthBenchProfessional": 48.1
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 73.9
        },
        "math": {
          "frontierMathV2Tiers13": 47.6,
          "frontierMathV2Tier4": 27.1
        },
        "korean": {}
      }
    },
    {
      "slug": "kimi-2-6",
      "canonicalModelKey": "kimi-2-6",
      "model": "Kimi K2.6",
      "creator": "Moonshot AI",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": 59.09,
      "provisionalDisplayScore": 62,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 49.22,
        "upper": 68.96
      },
      "rankingEligible": true,
      "overallRank": 81,
      "url": "https://benchlm.ai/models/kimi-2-6",
      "markdownUrl": "https://benchlm.ai/md/models/kimi-2-6.md",
      "id": 44,
      "releaseDate": "2026-04-20",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "kimi-2-6",
        "familyName": "Kimi K2.6",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "kimi-2-6",
        "relatedModelKeys": [
          "kimi-k2-5"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "kimi-k2-5"
      },
      "scores": {
        "displayScore": 62,
        "overallScore": 62,
        "rawOverallScore": 61,
        "verifiedDisplayScore": 62,
        "displayCategoryScores": {
          "agentic": 66.6,
          "coding": 60,
          "reasoning": null,
          "multimodalGrounded": 61.9,
          "knowledge": 51.2,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 71.7
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 66.6,
          "coding": 60,
          "reasoning": null,
          "multimodalGrounded": 61.9,
          "knowledge": 49.2,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 71.7
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 81,
        "categoryRanks": {
          "agentic": 95,
          "coding": 76,
          "multimodalGrounded": 18,
          "knowledge": 52,
          "math": 1
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": true
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 29,
        "verifiedBenchmarkCount": 29,
        "rankableBenchmarkCount": 29,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 3
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 66.7,
          "browseComp": 83.2,
          "osWorldVerified": 73.1,
          "toolathlon": 50,
          "mcpAtlas": 55.9,
          "clawEval": 62.3,
          "deepSearchQa": 92.5,
          "wideResearch": 80.8,
          "aaAgenticIndex": 31.16,
          "tau2Bench": 95.9,
          "gdpvalAaNormalized": 34.5,
          "gdpvalAa": 1189,
          "apexAgentsAa": 28.5,
          "gertLabs": 56.82,
          "researchClawBench": 18,
          "osWorld2": 4.6
        },
        "coding": {
          "sweVerified": 80.2,
          "liveCodeBenchV6": 89.6,
          "swePro": 58.6,
          "sweMultilingual": 76.7,
          "sciCode": 52.2,
          "terminalBench2": 66.7,
          "vibeCodeBench": 37.891,
          "cursorBench31": 47.6,
          "aaCodingIndex": 61.77,
          "aaSciCode": 53.5
        },
        "reasoning": {
          "lcr": 76.7,
          "critpt": 8
        },
        "multimodalGrounded": {
          "mmmuPro": 79.4,
          "mmmuProPython": 80.1,
          "charxiv": 80.4,
          "mathVision": 87.4,
          "vStar": 96.9,
          "aaMmmuPro": 79.4,
          "designArenaWebsite": 1289
        },
        "knowledge": {
          "gpqa": 90.5,
          "gpqaDiamond": 90.5,
          "hle": 34.7,
          "artificialAnalysis": 45.14,
          "aaGpqaDiamond": 91.1,
          "aaHle": 37.5,
          "aaOmniscienceIndex": 5.3,
          "omniscienceAccuracy": 32.6,
          "omniscienceHallucinationRate": 40.5
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 76
        },
        "math": {
          "aime2026": 96.4,
          "hmmtFeb2026": 92.7,
          "mmAnswerBench": 86,
          "frontierMathV2Tiers13": 38.966,
          "frontierMathV2Tier4": 14.58
        }
      }
    },
    {
      "slug": "glm-5-1",
      "canonicalModelKey": "glm-5-1",
      "model": "GLM-5.1",
      "creator": "Z.AI",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "203K",
      "contextWindowTokens": 203000,
      "displayScore": 66.68,
      "provisionalDisplayScore": 61,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 55.9,
        "upper": 77.46
      },
      "rankingEligible": true,
      "overallRank": 29,
      "url": "https://benchlm.ai/models/glm-5-1",
      "markdownUrl": "https://benchlm.ai/md/models/glm-5-1.md",
      "id": 43,
      "releaseDate": "2026-04-07",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "glm-5",
        "familyName": "GLM-5",
        "variantType": "snapshot",
        "snapshotLabel": "5.1",
        "baseFamilyModelKey": "glm-5",
        "relatedModelKeys": [
          "glm-5",
          "glm-5-reasoning",
          "glm-5-turbo"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "glm-5"
      },
      "scores": {
        "displayScore": 61,
        "overallScore": 61,
        "rawOverallScore": 61,
        "verifiedDisplayScore": 61,
        "displayCategoryScores": {
          "agentic": 49.1,
          "coding": 63.3,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 75.4,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 64.5
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 49.1,
          "coding": 63.3,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 75.4,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 64.5
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 29,
        "categoryRanks": {
          "agentic": 90,
          "coding": 28,
          "math": 3
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": true
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 17,
        "verifiedBenchmarkCount": 17,
        "rankableBenchmarkCount": 17,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 63.5,
          "browseComp": 68,
          "tau3Bench": 70.6,
          "mcpAtlas": 71.8,
          "cyberGym": 68.7,
          "clawEval": 62.3,
          "aaAgenticIndex": 30.56,
          "tau2Bench": 97.7,
          "gdpvalAaNormalized": 37.7,
          "gertLabs": 60.11,
          "gdpvalAa": 1254,
          "researchClawBench": 18.2
        },
        "coding": {
          "swePro": 58.4,
          "nl2Repo": 42.7,
          "sweRebench": 62.7,
          "vibeCodeBench": 31.456,
          "aaCodingIndex": 55.78,
          "aaSciCode": 43.8,
          "openHarmonyBench": 52.3
        },
        "reasoning": {
          "lcr": 68,
          "critpt": 4.6
        },
        "multimodalGrounded": {
          "designArenaWebsite": 1290
        },
        "knowledge": {
          "gpqaDiamond": 86.2,
          "hle": 52.3,
          "artificialAnalysis": 40.97,
          "aaGpqaDiamond": 86.8,
          "aaHle": 30.1,
          "aaOmniscienceIndex": 0.9,
          "omniscienceAccuracy": 23.7,
          "omniscienceHallucinationRate": 29.9
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 76.3
        },
        "math": {
          "aime2026": 95.3,
          "hmmtNov2025": 94,
          "hmmtFeb2026": 82.6,
          "mmAnswerBench": 83.8,
          "frontierMathV2Tiers13": 33.448,
          "frontierMathV2Tier4": 12.5
        }
      }
    },
    {
      "slug": "claude-sonnet-4-6",
      "canonicalModelKey": "claude-sonnet-4-6",
      "model": "Claude Sonnet 4.6",
      "creator": "Anthropic",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": 64.47,
      "provisionalDisplayScore": 60,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 51.06,
        "upper": 77.88
      },
      "rankingEligible": true,
      "overallRank": 41,
      "url": "https://benchlm.ai/models/claude-sonnet-4-6",
      "markdownUrl": "https://benchlm.ai/md/models/claude-sonnet-4-6.md",
      "id": 27,
      "releaseDate": "2026-02-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "claude-sonnet-4-6",
        "familyName": "Claude Sonnet 4.6",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "claude-sonnet-4-6",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 60,
        "overallScore": 60,
        "rawOverallScore": 59,
        "verifiedDisplayScore": 60,
        "displayCategoryScores": {
          "agentic": 54.5,
          "coding": 66.7,
          "reasoning": null,
          "multimodalGrounded": 46.7,
          "knowledge": 76,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 49.4
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 54.2,
          "coding": 66.7,
          "reasoning": null,
          "multimodalGrounded": 46.7,
          "knowledge": 76,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 49.4
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 41,
        "categoryRanks": {
          "agentic": 72,
          "coding": 42,
          "knowledge": 20
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 18,
        "verifiedBenchmarkCount": 18,
        "rankableBenchmarkCount": 18,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 3
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 59.1,
          "osWorldVerified": 72.1,
          "clawEval": 67.8,
          "cyberGym": 65.2,
          "tau2Bench": 79.5,
          "gertLabs": 62.92,
          "osWorld2": 8.3,
          "jobBench": 36.9
        },
        "coding": {
          "sweVerified": 79.6,
          "sweRebench": 60.7,
          "reactNativeEvals": 80.6,
          "vibeCodeBench": 51.476,
          "cursorBench31": 48.8,
          "aaSciCode": 46.9,
          "frontierCode": 24.3
        },
        "reasoning": {
          "lcr": 62.3,
          "critpt": 0.9
        },
        "multimodalGrounded": {
          "charxiv": 77.4,
          "aaMmmuPro": 70.6,
          "designArenaWebsite": 1303
        },
        "knowledge": {
          "gpqa": 89.9,
          "superGpqa": 95,
          "mmluPro": 79.2,
          "hle": 49,
          "artificialAnalysis": 36.8,
          "aaGpqaDiamond": 79.9,
          "aaHle": 13.3,
          "aaOmniscienceIndex": -3.5,
          "omniscienceAccuracy": 38.6,
          "omniscienceHallucinationRate": 68.5
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 41.2
        },
        "math": {
          "frontierMathV2Tiers13": 32.4,
          "frontierMathV2Tier4": 8.3
        },
        "korean": {}
      }
    },
    {
      "slug": "inkling",
      "canonicalModelKey": "inkling",
      "model": "Inkling",
      "creator": "Thinking Machines Lab",
      "sourceType": "Open Weight",
      "reasoningType": "Hybrid",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 66.51,
      "provisionalDisplayScore": 58,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 59.05,
        "upper": 73.97
      },
      "rankingEligible": true,
      "overallRank": 31,
      "url": "https://benchlm.ai/models/inkling",
      "markdownUrl": "https://benchlm.ai/md/models/inkling.md",
      "id": 284,
      "releaseDate": "2026-07-15",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "inkling",
        "familyName": "Inkling",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "inkling",
        "relatedModelKeys": [
          "inkling-small"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 58,
        "overallScore": 58,
        "rawOverallScore": 57,
        "verifiedDisplayScore": 58,
        "displayCategoryScores": {
          "agentic": 57.6,
          "coding": 51.2,
          "reasoning": null,
          "multimodalGrounded": 42.9,
          "knowledge": 66.6,
          "multilingual": null,
          "instructionFollowing": 88.6,
          "math": 79
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 57.6,
          "coding": 50.4,
          "reasoning": null,
          "multimodalGrounded": 42.2,
          "knowledge": 66.6,
          "multilingual": null,
          "instructionFollowing": 88.6,
          "math": 79
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 31,
        "categoryRanks": {
          "agentic": 116,
          "coding": 117,
          "multimodalGrounded": 28,
          "knowledge": 30,
          "instructionFollowing": 15
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 16,
        "verifiedBenchmarkCount": 16,
        "rankableBenchmarkCount": 16,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 3
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 63.8,
          "browseComp": 77.1,
          "mcpAtlas": 74.1,
          "designArenaAgenticWebDev": 1257,
          "aaAgenticIndex": 34.13,
          "gdpvalAaNormalized": 36.7,
          "gdpvalAa": 1234,
          "aaBriefcaseElo": 844,
          "aaTau3Banking": 29.1,
          "aaEnterpriseOpsGym": 38,
          "aaTerminalBench21": 55.1
        },
        "coding": {
          "sweVerified": 77.6,
          "swePro": 54.3,
          "terminalBench2": 63.8,
          "aaCodingIndex": 52.06,
          "aaSciCode": 46.1
        },
        "reasoning": {
          "lcr": 73.3,
          "critpt": 5.4
        },
        "multimodalGrounded": {
          "mmmuPro": 73.5,
          "charxiv": 82,
          "charxivNoTools": 78.1,
          "aaMmmuPro": 73.5,
          "designArenaWebsite": 1229
        },
        "knowledge": {
          "gpqa": 87.9,
          "gpqaDiamond": 87.9,
          "hle": 46,
          "hleNoTools": 30,
          "artificialAnalysis": 42.29,
          "aaGpqaDiamond": 87.2,
          "aaHle": 31.9,
          "aaOmniscienceIndex": 2,
          "omniscienceAccuracy": 41.6,
          "omniscienceHallucinationRate": 67.7,
          "aaOpennessIndex": 38.9
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifBench": 79.8
        },
        "math": {
          "aime2026": 97.1
        }
      }
    },
    {
      "slug": "minimax-m3",
      "canonicalModelKey": "minimax-m3",
      "model": "MiniMax M3",
      "creator": "MiniMax",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 68.17,
      "provisionalDisplayScore": 58,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 62.45,
        "upper": 73.89
      },
      "rankingEligible": true,
      "overallRank": 22,
      "url": "https://benchlm.ai/models/minimax-m3",
      "markdownUrl": "https://benchlm.ai/md/models/minimax-m3.md",
      "id": 241,
      "releaseDate": "2026-06-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "minimax-m3",
        "familyName": "MiniMax M3",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "minimax-m3",
        "relatedModelKeys": [
          "minimax-m2-7",
          "minimax-m2-5"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "minimax-m2-7"
      },
      "scores": {
        "displayScore": 58,
        "overallScore": 58,
        "rawOverallScore": 57,
        "verifiedDisplayScore": 58,
        "displayCategoryScores": {
          "agentic": 63.5,
          "coding": 57.6,
          "reasoning": null,
          "multimodalGrounded": 42.6,
          "knowledge": 63.9,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 63.5,
          "coding": 57.5,
          "reasoning": null,
          "multimodalGrounded": 41.3,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 22,
        "categoryRanks": {
          "agentic": 114,
          "coding": 91,
          "multimodalGrounded": 29,
          "knowledge": 34
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 16,
        "verifiedBenchmarkCount": 16,
        "rankableBenchmarkCount": 16,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 66,
          "browseComp": 83.52,
          "osWorldVerified": 70.06,
          "mcpAtlas": 74.2,
          "clawEval": 74.5,
          "aaAgenticIndex": 36.12,
          "tau2Bench": 88.9,
          "gdpvalAaNormalized": 44,
          "gdpvalAa": 1380,
          "gdpvalRubrics": 74.7,
          "bankerToolBench": 76.1,
          "researchClawBench": 19.8,
          "osWorld2": 4.6,
          "aaBriefcaseElo": 1107,
          "aaEnterpriseOpsGym": 32.1,
          "aaHarveyLab": 88.4,
          "terminalBenchHard": 42.4,
          "aaTerminalBench21": 65.2,
          "aaAutomationBench": 15.9,
          "aaTau3Banking": 15.3
        },
        "coding": {
          "sweVerified": 80.5,
          "swePro": 59,
          "terminalBench2": 66,
          "nl2Repo": 42.13,
          "aaCodingIndex": 58.57,
          "aaSciCode": 45.4,
          "vibeV2": 50.1,
          "svgBench": 63.7,
          "kernelBenchHard": 28.8,
          "openHarmonyBench": 48.4
        },
        "reasoning": {
          "lcr": 80.3,
          "critpt": 3.7
        },
        "multimodalGrounded": {
          "officeQaPro": 45.1,
          "omniDocBench15": 91.6,
          "mmmuPro": 78.1,
          "videoMmmu": 84.6,
          "videoMmeWithSub": 85.4,
          "designArenaWebsite": 1277,
          "aaMmmuPro": 78.6
        },
        "knowledge": {
          "artificialAnalysis": 45.4,
          "aaGpqaDiamond": 92.9,
          "aaHle": 39,
          "aaOmniscienceIndex": 1.4,
          "omniscienceAccuracy": 16.7,
          "omniscienceHallucinationRate": 18.4,
          "aaOpennessIndex": 33.3
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 82.9
        },
        "math": {
          "usamo2026": 85.71
        }
      }
    },
    {
      "slug": "qwen3-6-plus",
      "canonicalModelKey": "qwen3-6-plus",
      "model": "Qwen3.6 Plus",
      "creator": "Alibaba",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 64.56,
      "provisionalDisplayScore": 58,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 54.9,
        "upper": 74.23
      },
      "rankingEligible": true,
      "overallRank": 40,
      "url": "https://benchlm.ai/models/qwen3-6-plus",
      "markdownUrl": "https://benchlm.ai/md/models/qwen3-6-plus.md",
      "id": 55,
      "releaseDate": "2026-04-02",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "qwen3-6-plus",
        "familyName": "Qwen3.6 Plus",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "qwen3-6-plus",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 58,
        "overallScore": 58,
        "rawOverallScore": 57,
        "verifiedDisplayScore": 58,
        "displayCategoryScores": {
          "agentic": 52.5,
          "coding": 54.5,
          "reasoning": 54.4,
          "multimodalGrounded": 61.1,
          "knowledge": 55.8,
          "multilingual": 69.7,
          "instructionFollowing": 86.4,
          "math": 62.6
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 51.9,
          "coding": 53.6,
          "reasoning": 54.4,
          "multimodalGrounded": 61.1,
          "knowledge": 54.4,
          "multilingual": 69.7,
          "instructionFollowing": 86.4,
          "math": 62.6
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 40,
        "categoryRanks": {
          "agentic": 122,
          "coding": 66,
          "multimodalGrounded": 19,
          "knowledge": 47,
          "multilingual": 4,
          "instructionFollowing": 20,
          "math": 4
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": true,
          "multilingual": true,
          "instructionFollowing": true,
          "math": true
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 39,
        "verifiedBenchmarkCount": 39,
        "rankableBenchmarkCount": 39,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 4
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 61.6,
          "clawEval": 58.8,
          "qwenClawBench": 57.2,
          "tau3Bench": 70.7,
          "vitaBench": 44.3,
          "deepPlanning": 41.5,
          "toolathlon": 39.8,
          "mcpAtlas": 48.2,
          "mcpTasks": 74.1,
          "wideResearch": 74.3,
          "aaAgenticIndex": 29,
          "tau2Bench": 97.7,
          "gdpvalAaNormalized": 31.8,
          "gdpvalAa": 1137,
          "gertLabs": 50.6,
          "researchClawBench": 18
        },
        "coding": {
          "sweVerified": 78.8,
          "swePro": 56.6,
          "sweMultilingual": 73.8,
          "liveCodeBenchV6": 87.1,
          "vibeCodeBench": 25.564,
          "aaCodingIndex": 54.53,
          "aaSciCode": 40.7
        },
        "reasoning": {
          "aiNeedle": 68.3,
          "longBenchV2": 62,
          "lcr": 72.3,
          "critpt": 2.9
        },
        "multimodalGrounded": {
          "mmmu": 86,
          "mmmuPro": 78.8,
          "mathVision": 88,
          "videoMmmu": 84,
          "screenSpotPro": 68.2,
          "charxiv": 81.5,
          "vStar": 96.9,
          "aaMmmuPro": 78,
          "designArenaWebsite": 1258
        },
        "knowledge": {
          "gpqa": 90.4,
          "superGpqa": 71.6,
          "mmluPro": 88.5,
          "mmluRedux": 94.5,
          "cEval": 93.3,
          "hle": 28.8,
          "artificialAnalysis": 40.49,
          "aaGpqaDiamond": 88.2,
          "aaHle": 27.8,
          "aaOmniscienceIndex": 0.9,
          "omniscienceAccuracy": 26.4,
          "omniscienceHallucinationRate": 34.6
        },
        "multilingual": {
          "mmluProX": 84.7,
          "nova63": 57.9
        },
        "instructionFollowing": {
          "ifeval": 94.3,
          "ifBench": 75.8,
          "aaIfBench": 75.2
        },
        "math": {
          "aime2026": 95.3,
          "hmmtFeb2025": 96.7,
          "hmmtNov2025": 94.6,
          "hmmtFeb2026": 87.8,
          "mmAnswerBench": 83.8,
          "frontierMathV2Tiers13": 26.207,
          "frontierMathV2Tier4": 8.333
        }
      }
    },
    {
      "slug": "inkling-small",
      "canonicalModelKey": "inkling-small",
      "model": "Inkling-Small",
      "creator": "Thinking Machines Lab",
      "sourceType": "Open Weight",
      "reasoningType": "Hybrid",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 63.52,
      "provisionalDisplayScore": 57,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 57.93,
        "upper": 69.12
      },
      "rankingEligible": true,
      "overallRank": 45,
      "url": "https://benchlm.ai/models/inkling-small",
      "markdownUrl": "https://benchlm.ai/md/models/inkling-small.md",
      "id": 297,
      "releaseDate": "2026-07-30",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "inkling",
        "familyName": "Inkling",
        "variantType": "small",
        "snapshotLabel": "Small",
        "baseFamilyModelKey": "inkling",
        "relatedModelKeys": [
          "inkling"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 57,
        "overallScore": 57,
        "rawOverallScore": 56,
        "verifiedDisplayScore": 57,
        "displayCategoryScores": {
          "agentic": 58.8,
          "coding": 55.1,
          "reasoning": 41.7,
          "multimodalGrounded": 43.1,
          "knowledge": 69.7,
          "multilingual": null,
          "instructionFollowing": 92.8,
          "math": 77
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 58.8,
          "coding": 55.1,
          "reasoning": 41.7,
          "multimodalGrounded": 43.1,
          "knowledge": 69.7,
          "multilingual": null,
          "instructionFollowing": 92.8,
          "math": 77
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 45,
        "categoryRanks": {
          "agentic": 112,
          "coding": 50,
          "multimodalGrounded": 27,
          "instructionFollowing": 5
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 18,
        "verifiedBenchmarkCount": 18,
        "rankableBenchmarkCount": 18,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 3
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 64.7,
          "browseComp": 77.4,
          "mcpAtlas": 79.6,
          "toolathlonVerified": 54.4,
          "aaAgenticIndex": 31.88,
          "gdpvalAaNormalized": 38.4,
          "gdpvalAa": 1268
        },
        "coding": {
          "sweVerified": 80.2,
          "swePro": 55.9,
          "terminalBench2": 64.7,
          "sciCode": 48.7,
          "aaCodingIndex": 52.95,
          "aaSciCode": 48.7
        },
        "reasoning": {
          "arcAgi2": 40.1,
          "critpt": 8.3,
          "lcr": 69.3
        },
        "multimodalGrounded": {
          "mmmuPro": 74,
          "charxiv": 81.3,
          "charxivNoTools": 77.4,
          "aaMmmuPro": 74
        },
        "knowledge": {
          "gpqa": 89.5,
          "gpqaDiamond": 89.5,
          "hle": 47.8,
          "hleNoTools": 31.6,
          "artificialAnalysis": 41.18,
          "aaGpqaDiamond": 89.5,
          "aaHle": 33.3,
          "aaOmniscienceIndex": -8.9,
          "omniscienceAccuracy": 33.2,
          "omniscienceHallucinationRate": 63
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifBench": 82.2
        },
        "math": {
          "aime2026": 95.5,
          "hmmtFeb2026": 90.2
        }
      }
    },
    {
      "slug": "glm-5",
      "canonicalModelKey": "glm-5",
      "model": "GLM-5",
      "creator": "Z.AI",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": 65.49,
      "provisionalDisplayScore": 56,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 53.78,
        "upper": 77.2
      },
      "rankingEligible": true,
      "overallRank": 35,
      "url": "https://benchlm.ai/models/glm-5",
      "markdownUrl": "https://benchlm.ai/md/models/glm-5.md",
      "id": 51,
      "releaseDate": "2026-03-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "glm-5",
        "familyName": "GLM-5",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "glm-5",
        "relatedModelKeys": [
          "glm-5-reasoning"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 56,
        "overallScore": 56,
        "rawOverallScore": 56,
        "verifiedDisplayScore": 56,
        "displayCategoryScores": {
          "agentic": 43.9,
          "coding": 61.6,
          "reasoning": 43,
          "multimodalGrounded": null,
          "knowledge": 77.1,
          "multilingual": 48.7,
          "instructionFollowing": 87.4,
          "math": 56.9
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 43.9,
          "coding": 61.6,
          "reasoning": 43,
          "multimodalGrounded": null,
          "knowledge": 77.1,
          "multilingual": 48.7,
          "instructionFollowing": 87.4,
          "math": 56.9
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 35,
        "categoryRanks": {
          "agentic": 41,
          "coding": 37,
          "knowledge": 19,
          "multilingual": 6,
          "instructionFollowing": 18,
          "math": 7
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": true,
          "multilingual": true,
          "instructionFollowing": true,
          "math": true
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 34,
        "verifiedBenchmarkCount": 34,
        "rankableBenchmarkCount": 34,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 4
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 56.2,
          "clawEval": 57.7,
          "qwenClawBench": 54.1,
          "tau3Bench": 65.6,
          "deepPlanning": 14.6,
          "toolathlon": 38,
          "mcpAtlas": 31.1,
          "mcpTasks": 60.8,
          "wideResearch": 69.8,
          "tau2Bench": 98.2,
          "cyberGym": 43.2,
          "apexAgentsAa": 14.5,
          "gertLabs": 50.99
        },
        "coding": {
          "sweVerified": 77.8,
          "sweVerifiedArcee": 72.8,
          "swePro": 55.1,
          "sweMultilingual": 73.3,
          "sweRebench": 62.8,
          "reactNativeEvals": 74.8,
          "aaSciCode": 46.2
        },
        "reasoning": {
          "longBenchV2": 60.8,
          "aiNeedle": 63.3,
          "lcr": 70.7,
          "critpt": 2
        },
        "multimodalGrounded": {
          "designArenaWebsite": 1266
        },
        "knowledge": {
          "gpqa": 86,
          "gpqaDiamond": 86,
          "superGpqa": 66.8,
          "mmluPro": 85.7,
          "mmluProArcee": 85.8,
          "hle": 50.4,
          "artificialAnalysis": 40.55,
          "aaGpqaDiamond": 82,
          "aaHle": 29.3,
          "aaOmniscienceIndex": 0.3,
          "omniscienceAccuracy": 26.3,
          "omniscienceHallucinationRate": 35.3
        },
        "multilingual": {
          "mmluProX": 83.1,
          "nova63": 55.1
        },
        "instructionFollowing": {
          "ifeval": 92.6,
          "aaIfBench": 72.3
        },
        "math": {
          "aime2026": 95.8,
          "aime2025Arcee": 93.3,
          "hmmtFeb2025": 97.5,
          "hmmtNov2025": 96.9,
          "hmmtFeb2026": 86.4,
          "mmAnswerBench": 82.5,
          "frontierMathV2Tiers13": 16.434,
          "frontierMathV2Tier4": 2.1
        }
      }
    },
    {
      "slug": "gpt-5-4-mini",
      "canonicalModelKey": "gpt-5-4-mini",
      "model": "GPT-5.4 mini",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "400K",
      "contextWindowTokens": 400000,
      "displayScore": 56.69,
      "provisionalDisplayScore": 55,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 45.17,
        "upper": 68.2
      },
      "rankingEligible": true,
      "overallRank": 96,
      "url": "https://benchlm.ai/models/gpt-5-4-mini",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-5-4-mini.md",
      "id": 81,
      "releaseDate": "2026-03-17",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-5-4",
        "familyName": "GPT-5.4",
        "variantType": "mini",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-5-4",
        "relatedModelKeys": [
          "gpt-5-4",
          "gpt-5-4-pro",
          "gpt-5-4-nano"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": "gpt-5-mini"
      },
      "scores": {
        "displayScore": 55,
        "overallScore": 55,
        "rawOverallScore": 55,
        "verifiedDisplayScore": 55,
        "displayCategoryScores": {
          "agentic": 55.1,
          "coding": 52.6,
          "reasoning": null,
          "multimodalGrounded": 51.8,
          "knowledge": 57,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 44.6
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 55.1,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": 51.8,
          "knowledge": 56.7,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 44.6
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": false,
        "overallRank": 96,
        "categoryRanks": {
          "agentic": 121,
          "coding": 100,
          "knowledge": 46
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 12,
        "verifiedBenchmarkCount": 12,
        "rankableBenchmarkCount": 12,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 3
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 60,
          "osWorldVerified": 72.1,
          "mcpAtlas": 57.7,
          "toolathlon": 42.9,
          "tau2Bench": 93.4,
          "aaAgenticIndex": 31.54,
          "apexAgentsAa": 28.2,
          "gdpvalAaNormalized": 33.3,
          "gdpvalAa": 1167
        },
        "coding": {
          "vibeCodeBench": 47.969,
          "aaCodingIndex": 56.08,
          "aaSciCode": 49.9,
          "frontierCode": 27
        },
        "reasoning": {
          "lcr": 73,
          "critpt": 10
        },
        "multimodalGrounded": {
          "mmmuPro": 76.6,
          "mmmuProPython": 78,
          "aaMmmuPro": 73.3
        },
        "knowledge": {
          "gpqa": 88,
          "hle": 41.5,
          "hleNoTools": 28.2,
          "artificialAnalysis": 40.94,
          "aaGpqaDiamond": 87.5,
          "aaHle": 28.1,
          "aaOmniscienceIndex": -18.9,
          "omniscienceAccuracy": 37.5,
          "omniscienceHallucinationRate": 90.2
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 73.3
        },
        "math": {
          "frontierMathV2Tiers13": 28.28,
          "frontierMathV2Tier4": 2.08
        }
      }
    },
    {
      "slug": "claude-opus-4-5",
      "canonicalModelKey": "claude-opus-4-5",
      "model": "Claude Opus 4.5",
      "creator": "Anthropic",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": 63.51,
      "provisionalDisplayScore": 55,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 49.13,
        "upper": 77.89
      },
      "rankingEligible": true,
      "overallRank": 46,
      "url": "https://benchlm.ai/models/claude-opus-4-5",
      "markdownUrl": "https://benchlm.ai/md/models/claude-opus-4-5.md",
      "id": 36,
      "releaseDate": "2025-11-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "claude-opus-4-5",
        "familyName": "Claude Opus 4.5",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "claude-opus-4-5",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 55,
        "overallScore": 55,
        "rawOverallScore": 54,
        "verifiedDisplayScore": 55,
        "displayCategoryScores": {
          "agentic": 48.1,
          "coding": 57,
          "reasoning": 77,
          "multimodalGrounded": 13.5,
          "knowledge": 58,
          "multilingual": 82.9,
          "instructionFollowing": 56.4,
          "math": 58.4
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 48.1,
          "coding": 56.4,
          "reasoning": 77,
          "multimodalGrounded": 13.5,
          "knowledge": 56.6,
          "multilingual": 82.9,
          "instructionFollowing": 56.4,
          "math": 58.4
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 46,
        "categoryRanks": {
          "agentic": 124,
          "coding": 34,
          "multimodalGrounded": 32,
          "knowledge": 44,
          "multilingual": 2,
          "instructionFollowing": 33,
          "math": 6
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": true,
          "multilingual": true,
          "instructionFollowing": true,
          "math": true
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 41,
        "verifiedBenchmarkCount": 41,
        "rankableBenchmarkCount": 41,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 4
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 59.3,
          "osWorldVerified": 66.3,
          "osWorld": 66.3,
          "clawEval": 59.6,
          "qwenClawBench": 52.3,
          "tau3Bench": 70.2,
          "vitaBench": 23.3,
          "deepPlanning": 26.4,
          "toolathlon": 43.5,
          "mcpAtlas": 42.3,
          "mcpTasks": 71.8,
          "wideResearch": 76.4,
          "cyberGym": 50.6,
          "tau2Bench": 86.3,
          "gertLabs": 64.23,
          "jobBench": 32.3
        },
        "coding": {
          "sweVerified": 80.9,
          "liveCodeBenchV6": 84.8,
          "swePro": 57.1,
          "sweMultilingual": 77.5,
          "nl2Repo": 43.2,
          "aaSciCode": 47
        },
        "reasoning": {
          "longBenchV2": 64.4,
          "aiNeedle": 74,
          "lcr": 67.3,
          "critpt": 0.3
        },
        "multimodalGrounded": {
          "mmmuPro": 70.6,
          "mathVision": 74.3,
          "charxiv": 68.5,
          "videoMmmu": 84.4,
          "screenSpotPro": 45.7,
          "vStar": 67,
          "aaMmmuPro": 71.2,
          "designArenaWebsite": 1265
        },
        "knowledge": {
          "gpqa": 87,
          "superGpqa": 70.6,
          "mmluPro": 89.5,
          "mmluRedux": 96.6,
          "cEval": 92.2,
          "hle": 30.8,
          "artificialAnalysis": 35.57,
          "aaGpqaDiamond": 81,
          "aaHle": 13.2,
          "aaOmniscienceIndex": -4.1,
          "omniscienceAccuracy": 40.9,
          "omniscienceHallucinationRate": 76.2,
          "aaMmluPro": 88.9
        },
        "multilingual": {
          "mmluProX": 85.7,
          "nova63": 56.7
        },
        "instructionFollowing": {
          "ifeval": 90.9,
          "ifBench": 58,
          "aaIfBench": 43
        },
        "math": {
          "aime2026": 95.1,
          "hmmtFeb2025": 92.9,
          "hmmtNov2025": 93.3,
          "hmmtFeb2026": 85.3,
          "mmAnswerBench": 84,
          "frontierMathV2Tiers13": 20.69,
          "frontierMathV2Tier4": 4.167
        }
      }
    },
    {
      "slug": "qwen3-5-397b",
      "canonicalModelKey": "qwen3-5-397b",
      "model": "Qwen3.5 397B",
      "creator": "Alibaba",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 57.75,
      "provisionalDisplayScore": 54,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 46.24,
        "upper": 69.27
      },
      "rankingEligible": true,
      "overallRank": 89,
      "url": "https://benchlm.ai/models/qwen3-5-397b",
      "markdownUrl": "https://benchlm.ai/md/models/qwen3-5-397b.md",
      "id": 57,
      "releaseDate": "2026-02-16",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "qwen3-5-397b",
        "familyName": "Qwen3.5 397B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "qwen3-5-397b",
        "relatedModelKeys": [
          "qwen3-5-397b-reasoning"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 54,
        "overallScore": 54,
        "rawOverallScore": 53,
        "verifiedDisplayScore": 54,
        "displayCategoryScores": {
          "agentic": 32.4,
          "coding": 46,
          "reasoning": 65.7,
          "multimodalGrounded": 60.9,
          "knowledge": 53.5,
          "multilingual": 69.7,
          "instructionFollowing": 87.4,
          "math": 74.3
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 32.4,
          "coding": 46,
          "reasoning": 65.7,
          "multimodalGrounded": 60.9,
          "knowledge": 53.5,
          "multilingual": 69.7,
          "instructionFollowing": 87.4,
          "math": 74.3
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 89,
        "categoryRanks": {
          "agentic": 77,
          "coding": 57,
          "multimodalGrounded": 20,
          "knowledge": 49,
          "multilingual": 5,
          "instructionFollowing": 19
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": true,
          "multilingual": true,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 35,
        "verifiedBenchmarkCount": 35,
        "rankableBenchmarkCount": 35,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 4
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 52.5,
          "browseComp": 62,
          "clawEval": 56.8,
          "qwenClawBench": 51.8,
          "tau3Bench": 68.4,
          "vitaBench": 43.7,
          "deepPlanning": 37.6,
          "toolathlon": 36.3,
          "mcpAtlas": 46.1,
          "mcpTasks": 74.2,
          "wideResearch": 74,
          "tau2Bench": 95.6,
          "gertLabs": 46.76,
          "researchClawBench": 14.2,
          "aaAgenticIndex": 19.85,
          "apexAgentsAa": 15.3,
          "gdpvalAaNormalized": 23.3,
          "gdpvalAa": 966
        },
        "coding": {
          "sweVerified": 76.2,
          "liveCodeBenchV6": 83.6,
          "swePro": 50.9,
          "aaSciCode": 42,
          "aaCodingIndex": 48.21
        },
        "reasoning": {
          "longBenchV2": 63.2,
          "aiNeedle": 68.7,
          "lcr": 72.7,
          "critpt": 1.7
        },
        "multimodalGrounded": {
          "mmmuPro": 79,
          "mathVision": 88.6,
          "charxiv": 80.8,
          "videoMmmu": 84.7,
          "screenSpotPro": 65.6,
          "vStar": 95.8,
          "aaMmmuPro": 77.3
        },
        "knowledge": {
          "gpqa": 88.4,
          "superGpqa": 70.4,
          "mmluPro": 87.8,
          "mmluRedux": 94.9,
          "cEval": 93,
          "hle": 28.7,
          "artificialAnalysis": 34.26,
          "aaGpqaDiamond": 89.3,
          "aaHle": 29,
          "aaOmniscienceIndex": -30.7,
          "omniscienceAccuracy": 30.8,
          "omniscienceHallucinationRate": 88.9
        },
        "multilingual": {
          "mmluProX": 84.7,
          "nova63": 59.1
        },
        "instructionFollowing": {
          "ifeval": 92.6,
          "aaIfBench": 78.8
        },
        "math": {
          "aime2026": 93.3,
          "hmmtFeb2025": 94.8,
          "hmmtNov2025": 92.7,
          "hmmtFeb2026": 87.9,
          "mmAnswerBench": 80.9
        }
      }
    },
    {
      "slug": "muse-glimmer-30b",
      "canonicalModelKey": "muse-glimmer-30b",
      "model": "Muse Glimmer 30B",
      "creator": "Meta",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "131K",
      "contextWindowTokens": 131000,
      "displayScore": null,
      "provisionalDisplayScore": 54,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/muse-glimmer-30b",
      "markdownUrl": "https://benchlm.ai/md/models/muse-glimmer-30b.md",
      "id": 402,
      "releaseDate": "2026-08-10",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "muse-glimmer",
        "familyName": "Muse Glimmer",
        "variantType": "30b",
        "snapshotLabel": "30B",
        "baseFamilyModelKey": "muse-glimmer-30b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 54,
        "overallScore": 54,
        "rawOverallScore": 53,
        "verifiedDisplayScore": 54,
        "displayCategoryScores": {
          "agentic": 52.1,
          "coding": 44.2,
          "reasoning": null,
          "multimodalGrounded": 39.8,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": 83.6,
          "math": 76.4
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 52.1,
          "coding": 44.2,
          "reasoning": null,
          "multimodalGrounded": 39.8,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": 83.6,
          "math": 76.4
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": null,
        "categoryRanks": {
          "agentic": 55,
          "coding": 63,
          "multimodalGrounded": 30,
          "instructionFollowing": 23
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 13,
        "verifiedBenchmarkCount": 13,
        "rankableBenchmarkCount": 8,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "mcpAtlas": 75.5,
          "deepSearchQa": 74.6,
          "skillsBench": 44.3,
          "osWorldVerified": 65.9,
          "aaAgenticIndex": 22.94,
          "gdpvalAaNormalized": 22.8,
          "gdpvalAa": 955,
          "aaTau3Banking": 23.5,
          "aaEnterpriseOpsGym": 34.7
        },
        "coding": {
          "swePro": 51.2,
          "sweVerified": 76,
          "terminalBench21": 51.7,
          "sciCode": 43.6,
          "aaCodingIndex": 49,
          "aaSciCode": 43.6
        },
        "reasoning": {
          "lcr": 80,
          "critpt": 2.6
        },
        "multimodalGrounded": {
          "charxiv": 78.8,
          "screenSpotPro": 75.4,
          "omniDocBench15": 75.8,
          "mmmuPro": 74,
          "aaMmmuPro": 74.3
        },
        "knowledge": {
          "artificialAnalysis": 35.06,
          "aaGpqaDiamond": 83.5,
          "aaHle": 22,
          "aaOmniscienceIndex": -32.8,
          "omniscienceAccuracy": 27,
          "omniscienceHallucinationRate": 81.9,
          "aaOpennessIndex": 44.4
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifBench": 77
        },
        "math": {
          "aime2026": 94.7
        }
      }
    },
    {
      "slug": "mai-thinking-1",
      "canonicalModelKey": "mai-thinking-1",
      "model": "MAI-Thinking-1",
      "creator": "Microsoft",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": 50.52,
      "provisionalDisplayScore": 54,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 40.65,
        "upper": 60.39
      },
      "rankingEligible": true,
      "overallRank": 135,
      "url": "https://benchlm.ai/models/mai-thinking-1",
      "markdownUrl": "https://benchlm.ai/md/models/mai-thinking-1.md",
      "id": 250,
      "releaseDate": "2026-06-02",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "mai-thinking",
        "familyName": "MAI-Thinking",
        "variantType": "1",
        "snapshotLabel": null,
        "baseFamilyModelKey": "mai-thinking-1",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 54,
        "overallScore": 54,
        "rawOverallScore": 53,
        "verifiedDisplayScore": 54,
        "displayCategoryScores": {
          "agentic": 28.7,
          "coding": 44.5,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 80.6,
          "multilingual": null,
          "instructionFollowing": 97.8,
          "math": 73.5
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 28.7,
          "coding": 44.5,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 80.6,
          "multilingual": null,
          "instructionFollowing": 97.8,
          "math": 73.5
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 135,
        "categoryRanks": {
          "agentic": 62,
          "coding": 71,
          "knowledge": 14,
          "instructionFollowing": 1
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 13,
        "verifiedBenchmarkCount": 13,
        "rankableBenchmarkCount": 13,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 3
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 46
        },
        "coding": {
          "liveCodeBenchV6": 87.7,
          "sweVerified": 73.5,
          "swePro": 52.8,
          "terminalBench2": 46
        },
        "reasoning": {
          "graphwalksBfs128k": 90
        },
        "multimodalGrounded": {},
        "knowledge": {
          "gpqa": 84.2,
          "gpqaDiamond": 84.2,
          "mmluPro": 85,
          "simpleQa": 31
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifBench": 85
        },
        "math": {
          "aime2025": 97,
          "aime2026": 94.5,
          "hmmtFeb2026": 84.9
        }
      }
    },
    {
      "slug": "gemini-3-pro",
      "canonicalModelKey": "gemini-3-pro",
      "model": "Gemini 3 Pro",
      "creator": "Google",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "2M",
      "contextWindowTokens": 2000000,
      "displayScore": 67.25,
      "provisionalDisplayScore": 53,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 55.19,
        "upper": 79.3
      },
      "rankingEligible": true,
      "overallRank": 26,
      "url": "https://benchlm.ai/models/gemini-3-pro",
      "markdownUrl": "https://benchlm.ai/md/models/gemini-3-pro.md",
      "id": 34,
      "releaseDate": "2025-11-18",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gemini-3-pro",
        "familyName": "Gemini 3 Pro",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gemini-3-pro",
        "relatedModelKeys": [
          "gemini-3-pro-deep-think"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 53,
        "overallScore": 53,
        "rawOverallScore": 52,
        "verifiedDisplayScore": 53,
        "displayCategoryScores": {
          "agentic": null,
          "coding": 59.9,
          "reasoning": 34.2,
          "multimodalGrounded": 69.1,
          "knowledge": 68.2,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 56.2
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": 34.2,
          "multimodalGrounded": 69.1,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 56.2
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": false,
        "overallRank": 26,
        "categoryRanks": {
          "agentic": 24,
          "coding": 30,
          "multimodalGrounded": 13,
          "knowledge": 27
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 11,
        "verifiedBenchmarkCount": 11,
        "rankableBenchmarkCount": 11,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 87.1,
          "gertLabs": 63.23,
          "jobBench": 11.4
        },
        "coding": {
          "vibeCodeBench": 14.3,
          "aaSciCode": 56.1,
          "aaLiveCodeBench": 91.7
        },
        "reasoning": {
          "arcAgi2": 31.1,
          "lcr": 73,
          "critpt": 9.1
        },
        "multimodalGrounded": {
          "mmmuPro": 81,
          "mathVision": 86.6,
          "videoMmmu": 87.6,
          "screenSpotPro": 72.7,
          "charxiv": 81.4,
          "vStar": 88,
          "aaMmmuPro": 80.2
        },
        "knowledge": {
          "artificialAnalysis": 40.61,
          "aaGpqaDiamond": 90.8,
          "aaHle": 39.7,
          "aaOmniscienceIndex": 15.3,
          "omniscienceAccuracy": 55.8,
          "omniscienceHallucinationRate": 91.5,
          "aaMmluPro": 89.8
        },
        "multilingual": {
          "aaGlobalMmluLite": 92.2
        },
        "instructionFollowing": {
          "aaIfBench": 70.4
        },
        "math": {
          "frontierMathV2Tiers13": 37.6,
          "frontierMathV2Tier4": 18.75
        }
      }
    },
    {
      "slug": "gpt-5-2",
      "canonicalModelKey": "gpt-5-2",
      "model": "GPT-5.2",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "400K",
      "contextWindowTokens": 400000,
      "displayScore": 57.95,
      "provisionalDisplayScore": 52,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 49.94,
        "upper": 65.96
      },
      "rankingEligible": true,
      "overallRank": 87,
      "url": "https://benchlm.ai/models/gpt-5-2",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-5-2.md",
      "id": 26,
      "releaseDate": "2025-12-11",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-5-2",
        "familyName": "GPT-5.2",
        "variantType": "thinking",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-5-2",
        "relatedModelKeys": [
          "gpt-5-2-instant",
          "gpt-5-2-pro"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 52,
        "overallScore": 52,
        "rawOverallScore": 51,
        "verifiedDisplayScore": 52,
        "displayCategoryScores": {
          "agentic": 22.2,
          "coding": 55.1,
          "reasoning": 52.3,
          "multimodalGrounded": 64.5,
          "knowledge": 83.7,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 58.4
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 22.2,
          "coding": 54.1,
          "reasoning": 52.3,
          "multimodalGrounded": 64.5,
          "knowledge": 83.7,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 58.4
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 87,
        "categoryRanks": {
          "agentic": 96,
          "coding": 53,
          "multimodalGrounded": 17,
          "knowledge": 8
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 14,
        "verifiedBenchmarkCount": 14,
        "rankableBenchmarkCount": 14,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 3
      },
      "benchmarks": {
        "agentic": {
          "browseComp": 65.8,
          "osWorldVerified": 47.3,
          "tau2Bench": 84.8,
          "gertLabs": 46.54,
          "jobBench": 34.3
        },
        "coding": {
          "sweVerified": 80,
          "swePro": 55.6,
          "vibeCodeBench": 53.499,
          "aaSciCode": 52.1
        },
        "reasoning": {
          "arcAgi2": 52.9,
          "lcr": 79.3,
          "critpt": 11.6
        },
        "multimodalGrounded": {
          "mmmuPro": 79.5,
          "mathVision": 83,
          "charxiv": 82.1,
          "vStar": 75.9,
          "designArenaWebsite": 1213
        },
        "knowledge": {
          "gpqa": 92.4,
          "artificialAnalysis": 43.34,
          "aaGpqaDiamond": 90.3,
          "aaHle": 37.7,
          "aaOmniscienceIndex": -0.9,
          "omniscienceAccuracy": 44.3,
          "omniscienceHallucinationRate": 81.2
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 75.4
        },
        "math": {
          "aaAime2025": 99,
          "frontierMathV2Tiers13": 40.7,
          "frontierMathV2Tier4": 18.8
        },
        "korean": {}
      }
    },
    {
      "slug": "qwen3-6-27b",
      "canonicalModelKey": "qwen3-6-27b",
      "model": "Qwen3.6-27B",
      "creator": "Alibaba",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": 53.57,
      "provisionalDisplayScore": 50,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 42.05,
        "upper": 65.08
      },
      "rankingEligible": true,
      "overallRank": 113,
      "url": "https://benchlm.ai/models/qwen3-6-27b",
      "markdownUrl": "https://benchlm.ai/md/models/qwen3-6-27b.md",
      "id": 63,
      "releaseDate": "2026-04-21",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "qwen3-6-27b",
        "familyName": "Qwen3.6-27B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "qwen3-6-27b",
        "relatedModelKeys": [
          "qwen3-6-35b-a3b",
          "qwen3-5-27b"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 50,
        "overallScore": 50,
        "rawOverallScore": 50,
        "verifiedDisplayScore": 50,
        "displayCategoryScores": {
          "agentic": 48.5,
          "coding": 44.9,
          "reasoning": null,
          "multimodalGrounded": 45.9,
          "knowledge": 46.7,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 72.9
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 48.5,
          "coding": 43.1,
          "reasoning": null,
          "multimodalGrounded": 45.9,
          "knowledge": 46.7,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 72.9
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 113,
        "categoryRanks": {
          "agentic": 130,
          "coding": 103,
          "multimodalGrounded": 25,
          "knowledge": 54
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 36,
        "verifiedBenchmarkCount": 36,
        "rankableBenchmarkCount": 36,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 3
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 59.3,
          "clawEval": 72.4,
          "qwenClawBench": 53.4,
          "qwenWebBench": 1487,
          "androidWorld": 70.3,
          "aaAgenticIndex": 27.51,
          "tau2Bench": 94.2,
          "gdpvalAaNormalized": 31.9,
          "gdpvalAa": 1138,
          "gertLabs": 54.84
        },
        "coding": {
          "sweVerified": 77.2,
          "sweMultilingual": 71.3,
          "swePro": 53.5,
          "terminalBench2": 59.3,
          "liveCodeBench": 83.9,
          "nl2Repo": 36.2,
          "aaCodingIndex": 53.72,
          "aaSciCode": 39.8
        },
        "reasoning": {
          "lcr": 73.3,
          "critpt": 1.1
        },
        "multimodalGrounded": {
          "mmmu": 82.9,
          "mmmuPro": 75.8,
          "realWorldQa": 84.1,
          "dynaMath": 85.6,
          "mStar": 81.4,
          "simpleVqa": 56.1,
          "charxiv": 78.4,
          "ccOcr": 81.2,
          "countBench": 97.8,
          "refcocoAvg": 92.5,
          "erqa": 62.5,
          "videoMmeWithSub": 87.7,
          "videoMmmu": 84.4,
          "mlvuAvg": 86.6,
          "vStar": 94.7,
          "aaMmmuPro": 74.6
        },
        "knowledge": {
          "mmluPro": 86.2,
          "mmluRedux": 93.5,
          "superGpqa": 66,
          "cEval": 91.4,
          "gpqa": 87.8,
          "hle": 24,
          "artificialAnalysis": 37.7,
          "aaGpqaDiamond": 84.2,
          "aaHle": 23.1,
          "aaOmniscienceIndex": -20,
          "omniscienceAccuracy": 19.6,
          "omniscienceHallucinationRate": 49.3
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 67.6
        },
        "math": {
          "hmmtFeb2025": 93.8,
          "hmmtNov2025": 90.7,
          "hmmtFeb2026": 84.3,
          "mmAnswerBench": 80.8,
          "aime2026": 94.1
        }
      }
    },
    {
      "slug": "ling-3-0-flash",
      "canonicalModelKey": "ling-3-0-flash",
      "model": "Ling 3.0 Flash",
      "creator": "InclusionAI",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": 53.62,
      "provisionalDisplayScore": 50,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 42.1,
        "upper": 65.13
      },
      "rankingEligible": true,
      "overallRank": 112,
      "url": "https://benchlm.ai/models/ling-3-0-flash",
      "markdownUrl": "https://benchlm.ai/md/models/ling-3-0-flash.md",
      "id": 294,
      "releaseDate": "2026-07-23",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "ling-3-0",
        "familyName": "Ling 3.0",
        "variantType": "flash",
        "snapshotLabel": null,
        "baseFamilyModelKey": "ling-3-0-flash",
        "relatedModelKeys": [
          "ling-3-0-flash-fin",
          "ling-3-0-flash-fp8",
          "ling-2-6-flash"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "ling-2-6-flash"
      },
      "scores": {
        "displayScore": 50,
        "overallScore": 50,
        "rawOverallScore": 50,
        "verifiedDisplayScore": 50,
        "displayCategoryScores": {
          "agentic": 53.9,
          "coding": 38.9,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 29.1,
          "multilingual": null,
          "instructionFollowing": 79.2,
          "math": 73.8
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 53.9,
          "coding": 38.9,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 29.1,
          "multilingual": null,
          "instructionFollowing": 79.2,
          "math": 73.8
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 112,
        "categoryRanks": {
          "agentic": 68,
          "coding": 56,
          "instructionFollowing": 25
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 16,
        "verifiedBenchmarkCount": 16,
        "rankableBenchmarkCount": 16,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 3
      },
      "benchmarks": {
        "agentic": {
          "terminalBench21": 57,
          "aaTau3Banking": 28,
          "mcpAtlas": 65.5,
          "skillsBench": 44.8,
          "bfclV4": 73,
          "gdpvalAa": 1107,
          "wideResearch": 73.6,
          "browseComp": 72.2,
          "draco": 70.4,
          "aaAgenticIndex": 29.31,
          "gdpvalAaNormalized": 30.3
        },
        "coding": {
          "swePro": 56.6,
          "sweMultilingual": 72.4,
          "terminalBench21": 57,
          "liveCodeBenchV5": 82.8,
          "sciCode": 41.24,
          "aaCodingIndex": 50.65,
          "aaSciCode": 41.1
        },
        "reasoning": {
          "lcr": 67,
          "critpt": 1.7
        },
        "multimodalGrounded": {},
        "knowledge": {
          "gpqa": 84.97,
          "gpqaDiamond": 84.97,
          "hle": 22.7,
          "artificialAnalysis": 37.82,
          "aaGpqaDiamond": 85.5,
          "aaHle": 23.7,
          "aaOmniscienceIndex": -17.9,
          "omniscienceAccuracy": 18.2,
          "omniscienceHallucinationRate": 44.1
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifBench": 74.5
        },
        "math": {
          "aime2026": 93.2,
          "hmmtFeb2026": 87,
          "imoAnswerBench": 83.7
        }
      }
    },
    {
      "slug": "kimi-k2-5",
      "canonicalModelKey": "kimi-k2-5",
      "model": "Kimi K2.5",
      "creator": "Moonshot AI",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": 58.74,
      "provisionalDisplayScore": 49,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 50.13,
        "upper": 67.34
      },
      "rankingEligible": true,
      "overallRank": 84,
      "url": "https://benchlm.ai/models/kimi-k2-5",
      "markdownUrl": "https://benchlm.ai/md/models/kimi-k2-5.md",
      "id": 53,
      "releaseDate": "2026-02-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "kimi-k2-5",
        "familyName": "Kimi K2.5",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "kimi-k2-5",
        "relatedModelKeys": [
          "kimi-k2-5-reasoning"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 49,
        "overallScore": 49,
        "rawOverallScore": 48,
        "verifiedDisplayScore": 49,
        "displayCategoryScores": {
          "agentic": 29.4,
          "coding": 53,
          "reasoning": 44.9,
          "multimodalGrounded": 61.2,
          "knowledge": 54.6,
          "multilingual": 38.2,
          "instructionFollowing": 91.2,
          "math": 62.5
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 29.4,
          "coding": 53,
          "reasoning": 44.9,
          "multimodalGrounded": 61.2,
          "knowledge": 54.6,
          "multilingual": 38.2,
          "instructionFollowing": 91.2,
          "math": 62.5
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 84,
        "categoryRanks": {
          "agentic": 117,
          "coding": 44,
          "knowledge": 48,
          "multilingual": 8,
          "instructionFollowing": 9,
          "math": 5
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": true,
          "multilingual": true,
          "instructionFollowing": true,
          "math": true
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 40,
        "verifiedBenchmarkCount": 40,
        "rankableBenchmarkCount": 40,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 4
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 50.8,
          "browseComp": 60.6,
          "clawEval": 52.3,
          "qwenClawBench": 54.3,
          "tau3Bench": 65.7,
          "deepSearchQa": 77.1,
          "deepPlanning": 14.4,
          "toolathlon": 27.8,
          "mcpAtlas": 29.5,
          "mcpTasks": 59.1,
          "wideResearch": 72.7,
          "tau2Bench": 95.9,
          "apexAgentsAa": 11.5,
          "gertLabs": 45.88,
          "researchClawBench": 14,
          "jobBench": 8.73,
          "aaAgenticIndex": 21.69,
          "gdpvalAaNormalized": 25.3,
          "gdpvalAa": 1006
        },
        "coding": {
          "sweVerified": 76.8,
          "sweVerifiedArcee": 70.8,
          "liveCodeBenchV6": 85,
          "swePro": 50.7,
          "sweMultilingual": 73,
          "sweRebench": 58.5,
          "reactNativeEvals": 77.2,
          "sciCode": 48.7,
          "aaSciCode": 49,
          "aaCodingIndex": 46.78
        },
        "reasoning": {
          "longBenchV2": 61,
          "lcr": 73,
          "critpt": 3.1
        },
        "multimodalGrounded": {
          "mmmuPro": 78.5,
          "videoMme": 87.4,
          "mmvu": 80.4,
          "videoMmmu": 86.6,
          "aaMmmuPro": 75.4,
          "designArenaWebsite": 1267
        },
        "knowledge": {
          "gpqa": 87.6,
          "gpqaDiamond": 87.6,
          "superGpqa": 69.2,
          "mmluPro": 87.1,
          "mmluProArcee": 87.1,
          "hle": 30.1,
          "artificialAnalysis": 36.02,
          "aaGpqaDiamond": 87.9,
          "aaHle": 30.7,
          "aaOmniscienceIndex": -7.3,
          "omniscienceAccuracy": 35.2,
          "omniscienceHallucinationRate": 65.7
        },
        "multilingual": {
          "mmluProX": 82.3,
          "nova63": 56
        },
        "instructionFollowing": {
          "ifeval": 93.9,
          "aaIfBench": 70.2
        },
        "math": {
          "aime2025": 96.1,
          "aime2026": 95.8,
          "aime2025Arcee": 96.3,
          "hmmtFeb2025": 95.4,
          "hmmtNov2025": 91.1,
          "hmmtFeb2026": 87.1,
          "mmAnswerBench": 81.8,
          "frontierMathV2Tiers13": 27.9,
          "frontierMathV2Tier4": 4.2
        }
      }
    },
    {
      "slug": "nemotron-3-ultra",
      "canonicalModelKey": "nemotron-3-ultra-500b",
      "model": "Nemotron 3 Ultra",
      "creator": "NVIDIA",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 45.68,
      "provisionalDisplayScore": 49,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 35.81,
        "upper": 55.55
      },
      "rankingEligible": true,
      "overallRank": 164,
      "url": "https://benchlm.ai/models/nemotron-3-ultra",
      "markdownUrl": "https://benchlm.ai/md/models/nemotron-3-ultra.md",
      "id": 71,
      "releaseDate": "2026-06-04",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "nemotron-3-ultra-500b",
        "familyName": "Nemotron 3 Ultra",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "nemotron-3-ultra-500b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 49,
        "overallScore": 48,
        "rawOverallScore": 47,
        "verifiedDisplayScore": 48,
        "displayCategoryScores": {
          "agentic": 29.4,
          "coding": 49.4,
          "reasoning": 53.4,
          "multimodalGrounded": null,
          "knowledge": 50.7,
          "multilingual": 47.4,
          "instructionFollowing": 91.9,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 27.4,
          "coding": 47.7,
          "reasoning": 53.4,
          "multimodalGrounded": null,
          "knowledge": 49.7,
          "multilingual": 47.4,
          "instructionFollowing": 91.9,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 164,
        "categoryRanks": {
          "agentic": 141,
          "coding": 140,
          "knowledge": 53,
          "multilingual": 7,
          "instructionFollowing": 7
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": true,
          "multilingual": true,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 17,
        "verifiedBenchmarkCount": 17,
        "rankableBenchmarkCount": 17,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 3
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 56.4,
          "pinchBench": 90,
          "browseComp": 44.4,
          "tau3Bench": 70.9,
          "gdpvalAaNormalized": 33.1,
          "hleWithTools": 37.4,
          "aaAgenticIndex": 27.5,
          "tau2Bench": 83.3,
          "gdpvalAa": 1162,
          "aaBriefcaseElo": 879,
          "aaEnterpriseOpsGym": 28.9,
          "aaHarveyLab": 81.7,
          "terminalBenchHard": 36.4,
          "aaAutomationBench": 5.7
        },
        "coding": {
          "sweVerified": 71.9,
          "sweMultilingual": 67.7,
          "liveCodeBenchV6": 89,
          "sciCode": 44.6,
          "terminalBench2": 56.4,
          "aaCodingIndex": 49.27,
          "aaSciCode": 39.9
        },
        "reasoning": {
          "lcr": 67,
          "critpt": 3.1,
          "longBenchV2": 61.9
        },
        "multimodalGrounded": {
          "designArenaWebsite": 1146
        },
        "knowledge": {
          "gpqa": 87,
          "gpqaDiamond": 87,
          "hle": 26.7,
          "hleNoTools": 26.7,
          "mmluPro": 86.8,
          "omniscienceAccuracy": 21.6,
          "artificialAnalysis": 38.32,
          "aaGpqaDiamond": 86.7,
          "aaHle": 28.4,
          "aaOmniscienceIndex": -0.4,
          "omniscienceHallucinationRate": 29.7,
          "aaOpennessIndex": 83.3
        },
        "multilingual": {
          "mmluProX": 83
        },
        "instructionFollowing": {
          "ifBench": 81.7,
          "aaIfBench": 81.4
        },
        "math": {}
      }
    },
    {
      "slug": "qwen3-5-27b",
      "canonicalModelKey": "qwen3-5-27b",
      "model": "Qwen3.5-27B",
      "creator": "Alibaba",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": 59.7,
      "provisionalDisplayScore": 48,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 49.77,
        "upper": 69.63
      },
      "rankingEligible": true,
      "overallRank": 74,
      "url": "https://benchlm.ai/models/qwen3-5-27b",
      "markdownUrl": "https://benchlm.ai/md/models/qwen3-5-27b.md",
      "id": 47,
      "releaseDate": "2026-03-04",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "qwen3-5-27b",
        "familyName": "Qwen3.5-27B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "qwen3-5-27b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 48,
        "overallScore": 48,
        "rawOverallScore": 47,
        "verifiedDisplayScore": 48,
        "displayCategoryScores": {
          "agentic": 14.9,
          "coding": 57.4,
          "reasoning": 41.1,
          "multimodalGrounded": null,
          "knowledge": 79,
          "multilingual": 36.8,
          "instructionFollowing": 93.9,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 14.9,
          "coding": 57.4,
          "reasoning": 41.1,
          "multimodalGrounded": null,
          "knowledge": 79,
          "multilingual": 36.8,
          "instructionFollowing": 93.9,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 74,
        "categoryRanks": {
          "agentic": 71,
          "coding": 52,
          "multimodalGrounded": 34,
          "knowledge": 16,
          "multilingual": 9,
          "instructionFollowing": 3
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": true,
          "multilingual": true,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 13,
        "verifiedBenchmarkCount": 13,
        "rankableBenchmarkCount": 13,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 3
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 41.6,
          "browseComp": 61,
          "osWorldVerified": 56.2,
          "tau2Bench": 93.9,
          "gertLabs": 39.41
        },
        "coding": {
          "sweVerified": 72.4,
          "sweRebench": 58.9,
          "aaSciCode": 39.5
        },
        "reasoning": {
          "longBenchV2": 60.6,
          "lcr": 72.3,
          "critpt": 0.9
        },
        "multimodalGrounded": {
          "mmmu": 82.3,
          "mmvu": 73.3,
          "mathVision": 86,
          "vStar": 93.7,
          "aaMmmuPro": 75
        },
        "knowledge": {
          "mmluPro": 86.1,
          "superGpqa": 65.6,
          "gpqa": 85.5,
          "artificialAnalysis": 34.6,
          "aaGpqaDiamond": 85.8,
          "aaHle": 23.9,
          "aaOmniscienceIndex": -44,
          "omniscienceAccuracy": 20.7,
          "omniscienceHallucinationRate": 81.5
        },
        "multilingual": {
          "mmluProX": 82.2
        },
        "instructionFollowing": {
          "ifeval": 95,
          "aaIfBench": 75.6
        },
        "math": {}
      }
    },
    {
      "slug": "qwen3-5-122b-a10b",
      "canonicalModelKey": "qwen3-5-122b-a10b",
      "model": "Qwen3.5-122B-A10B",
      "creator": "Alibaba",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": 59.49,
      "provisionalDisplayScore": 47,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 47.64,
        "upper": 71.34
      },
      "rankingEligible": true,
      "overallRank": 77,
      "url": "https://benchlm.ai/models/qwen3-5-122b-a10b",
      "markdownUrl": "https://benchlm.ai/md/models/qwen3-5-122b-a10b.md",
      "id": 37,
      "releaseDate": "2026-03-04",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "qwen3-5-122b-a10b",
        "familyName": "Qwen3.5-122B-A10B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "qwen3-5-122b-a10b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 47,
        "overallScore": 47,
        "rawOverallScore": 46,
        "verifiedDisplayScore": 47,
        "displayCategoryScores": {
          "agentic": 25,
          "coding": 51,
          "reasoning": 37.3,
          "multimodalGrounded": 46.1,
          "knowledge": 79.9,
          "multilingual": 36.8,
          "instructionFollowing": 89.8,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 25,
          "coding": 51,
          "reasoning": 37.3,
          "multimodalGrounded": 46.1,
          "knowledge": 79.9,
          "multilingual": 36.8,
          "instructionFollowing": 89.8,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 77,
        "categoryRanks": {
          "agentic": 70,
          "coding": 90,
          "knowledge": 15,
          "multilingual": 10,
          "instructionFollowing": 14
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": true,
          "multilingual": true,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 13,
        "verifiedBenchmarkCount": 13,
        "rankableBenchmarkCount": 13,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 3
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 49.4,
          "browseComp": 63.8,
          "osWorldVerified": 58,
          "tau2Bench": 93.6,
          "aaAgenticIndex": 21.27,
          "gdpvalAaNormalized": 24.4,
          "gdpvalAa": 987
        },
        "coding": {
          "sweVerified": 72,
          "aaCodingIndex": 45.71,
          "aaSciCode": 42
        },
        "reasoning": {
          "longBenchV2": 60.2,
          "lcr": 70.3,
          "critpt": 0.6
        },
        "multimodalGrounded": {
          "mmmu": 83.9,
          "mmvu": 74.7,
          "mathVision": 86.2,
          "charxiv": 77.2,
          "vStar": 93.2,
          "aaMmmuPro": 75
        },
        "knowledge": {
          "mmluPro": 86.7,
          "superGpqa": 67.1,
          "gpqa": 86.6,
          "artificialAnalysis": 32.85,
          "aaGpqaDiamond": 85.7,
          "aaHle": 25.2,
          "aaOmniscienceIndex": -41.5,
          "omniscienceAccuracy": 24.4,
          "omniscienceHallucinationRate": 87.1
        },
        "multilingual": {
          "mmluProX": 82.2
        },
        "instructionFollowing": {
          "ifeval": 93.4,
          "aaIfBench": 75.7
        },
        "math": {}
      }
    },
    {
      "slug": "qwen3-6-35b-a3b",
      "canonicalModelKey": "qwen3-6-35b-a3b",
      "model": "Qwen3.6-35B-A3B",
      "creator": "Alibaba",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": 51.32,
      "provisionalDisplayScore": 41,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 39.81,
        "upper": 62.84
      },
      "rankingEligible": true,
      "overallRank": 123,
      "url": "https://benchlm.ai/models/qwen3-6-35b-a3b",
      "markdownUrl": "https://benchlm.ai/md/models/qwen3-6-35b-a3b.md",
      "id": 76,
      "releaseDate": "2026-04-15",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "qwen3-6-35b-a3b",
        "familyName": "Qwen3.6-35B-A3B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "qwen3-6-35b-a3b",
        "relatedModelKeys": [
          "qwen3-6-plus",
          "qwen3-5-35b-a3b"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 41,
        "overallScore": 41,
        "rawOverallScore": 41,
        "verifiedDisplayScore": 41,
        "displayCategoryScores": {
          "agentic": 36.9,
          "coding": 23.8,
          "reasoning": null,
          "multimodalGrounded": 43.5,
          "knowledge": 42.8,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 71.6
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 36.9,
          "coding": 23.8,
          "reasoning": null,
          "multimodalGrounded": 43.5,
          "knowledge": 42.8,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 71.6
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 123,
        "categoryRanks": {
          "agentic": 88,
          "coding": 73,
          "multimodalGrounded": 26,
          "knowledge": 56
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 39,
        "verifiedBenchmarkCount": 39,
        "rankableBenchmarkCount": 39,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 3
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 51.5,
          "clawEval": 68.7,
          "qwenClawBench": 52.6,
          "qwenWebBench": 1397,
          "tau3Bench": 67.2,
          "vitaBench": 35.6,
          "deepPlanning": 25.9,
          "toolathlon": 26.9,
          "mcpAtlas": 62.8,
          "wideResearch": 60.1,
          "aaAgenticIndex": 21.62,
          "tau2Bench": 95.3,
          "gdpvalAaNormalized": 27.8,
          "gdpvalAa": 1056,
          "gertLabs": 42.65
        },
        "coding": {
          "sweVerified": 73.4,
          "sweMultilingual": 67.2,
          "swePro": 49.5,
          "terminalBench2": 51.5,
          "liveCodeBench": 80.4,
          "nl2Repo": 29.4,
          "aaCodingIndex": 41.88,
          "aaSciCode": 35.8
        },
        "reasoning": {
          "lcr": 66.7,
          "critpt": 0.3
        },
        "multimodalGrounded": {
          "mmmu": 81.7,
          "mmmuPro": 75.3,
          "realWorldQa": 85.3,
          "omniDocBench15": 89.9,
          "charxiv": 78,
          "simpleVqa": 58.9,
          "ccOcr": 81.9,
          "ai2dTest": 92.7,
          "refcocoAvg": 92,
          "odinw13": 50.8,
          "videoMmeWithSub": 86.6,
          "videoMmeNoSub": 82.5,
          "videoMmmu": 83.7,
          "mlvuAvg": 86.2,
          "aaMmmuPro": 75
        },
        "knowledge": {
          "mmluPro": 85.2,
          "superGpqa": 64.7,
          "cEval": 90,
          "gpqa": 86,
          "hle": 21.4,
          "artificialAnalysis": 32.13,
          "aaGpqaDiamond": 84.1,
          "aaHle": 22.2,
          "aaOmniscienceIndex": -22.2,
          "omniscienceAccuracy": 18.8,
          "omniscienceHallucinationRate": 50.5
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 64.4
        },
        "math": {
          "hmmtFeb2025": 90.7,
          "hmmtNov2025": 89.1,
          "hmmtFeb2026": 83.6,
          "mmAnswerBench": 78.9,
          "aime2026": 92.7
        }
      }
    },
    {
      "slug": "qwen3-5-35b-a3b",
      "canonicalModelKey": "qwen3-5-35b-a3b",
      "model": "Qwen3.5-35B-A3B",
      "creator": "Alibaba",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": 56.09,
      "provisionalDisplayScore": 40,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 44.13,
        "upper": 68.06
      },
      "rankingEligible": true,
      "overallRank": 100,
      "url": "https://benchlm.ai/models/qwen3-5-35b-a3b",
      "markdownUrl": "https://benchlm.ai/md/models/qwen3-5-35b-a3b.md",
      "id": 60,
      "releaseDate": "2026-03-04",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "qwen3-5-35b-a3b",
        "familyName": "Qwen3.5-35B-A3B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "qwen3-5-35b-a3b",
        "relatedModelKeys": [
          "qwen3-5-flash"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 40,
        "overallScore": 40,
        "rawOverallScore": 40,
        "verifiedDisplayScore": 40,
        "displayCategoryScores": {
          "agentic": 12.9,
          "coding": 46.3,
          "reasoning": 26,
          "multimodalGrounded": null,
          "knowledge": 77.8,
          "multilingual": 21.1,
          "instructionFollowing": 85.4,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 12.9,
          "coding": 46.3,
          "reasoning": 26,
          "multimodalGrounded": null,
          "knowledge": 77.8,
          "multilingual": 21.1,
          "instructionFollowing": 85.4,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 100,
        "categoryRanks": {
          "agentic": 82,
          "coding": 89,
          "multimodalGrounded": 35,
          "knowledge": 17,
          "multilingual": 11,
          "instructionFollowing": 22
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": true,
          "multilingual": true,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 13,
        "verifiedBenchmarkCount": 13,
        "rankableBenchmarkCount": 13,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 3
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 40.5,
          "browseComp": 61,
          "osWorldVerified": 54.5,
          "tau2Bench": 89.2,
          "gertLabs": 28.96
        },
        "coding": {
          "sweVerified": 69.2,
          "sweRebench": 53.7,
          "aaSciCode": 37.7
        },
        "reasoning": {
          "longBenchV2": 59,
          "lcr": 68.3,
          "critpt": 0.9
        },
        "multimodalGrounded": {
          "mmmu": 81.4,
          "mmvu": 72.3,
          "mathVision": 83.9,
          "vStar": 92.7,
          "aaMmmuPro": 72.7
        },
        "knowledge": {
          "mmluPro": 85.3,
          "superGpqa": 63.4,
          "gpqa": 84.2,
          "artificialAnalysis": 29.91,
          "aaGpqaDiamond": 84.5,
          "aaHle": 21,
          "aaOmniscienceIndex": -48.1,
          "omniscienceAccuracy": 20.1,
          "omniscienceHallucinationRate": 85.4
        },
        "multilingual": {
          "mmluProX": 81
        },
        "instructionFollowing": {
          "ifeval": 91.9,
          "aaIfBench": 72.5
        },
        "math": {}
      }
    },
    {
      "slug": "glm-4-7",
      "canonicalModelKey": "glm-4-7",
      "model": "GLM-4.7",
      "creator": "Z.AI",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": 60.62,
      "provisionalDisplayScore": 36,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 46.23,
        "upper": 75.01
      },
      "rankingEligible": true,
      "overallRank": 66,
      "url": "https://benchlm.ai/models/glm-4-7",
      "markdownUrl": "https://benchlm.ai/md/models/glm-4-7.md",
      "id": 48,
      "releaseDate": "2025-10-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "glm-4-7",
        "familyName": "GLM-4.7",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "glm-4-7",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 36,
        "overallScore": 36,
        "rawOverallScore": 36,
        "verifiedDisplayScore": 36,
        "displayCategoryScores": {
          "agentic": 11.5,
          "coding": 52.9,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 27.6,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 26
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 11.5,
          "coding": 51.9,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 27.6,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 26
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": true,
        "overallRank": 66,
        "categoryRanks": {
          "agentic": 60,
          "coding": 67
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 9,
        "verifiedBenchmarkCount": 9,
        "rankableBenchmarkCount": 9,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 41,
          "browseComp": 52,
          "vitaBench": 15.5,
          "aaAgenticIndex": 26.22,
          "tau2Bench": 95.9,
          "gertLabs": 39.95,
          "gdpvalAaNormalized": 33.3,
          "gdpvalAa": 1166
        },
        "coding": {
          "sweVerified": 73.8,
          "liveCodeBench": 84.9,
          "sweRebench": 58.7,
          "aaCodingIndex": 45.26,
          "aaSciCode": 45.1,
          "aaLiveCodeBench": 89.4
        },
        "reasoning": {
          "lcr": 68,
          "critpt": 1.7
        },
        "multimodalGrounded": {
          "designArenaWebsite": 1244
        },
        "knowledge": {
          "gpqa": 85.7,
          "mmluPro": 84.3,
          "hle": 24.8,
          "artificialAnalysis": 34.46,
          "aaGpqaDiamond": 85.9,
          "aaHle": 27.4,
          "aaOmniscienceIndex": -36.4,
          "omniscienceAccuracy": 29.3,
          "omniscienceHallucinationRate": 93
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 67.9
        },
        "math": {
          "aime2025": 95.7,
          "frontierMathV2Tiers13": 2.439,
          "frontierMathV2Tier4": 0
        }
      }
    },
    {
      "slug": "gpt-5-4-nano",
      "canonicalModelKey": "gpt-5-4-nano",
      "model": "GPT-5.4 nano",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "400K",
      "contextWindowTokens": 400000,
      "displayScore": 66.39,
      "provisionalDisplayScore": 36,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 55.08,
        "upper": 77.69
      },
      "rankingEligible": true,
      "overallRank": 32,
      "url": "https://benchlm.ai/models/gpt-5-4-nano",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-5-4-nano.md",
      "id": 127,
      "releaseDate": "2026-03-17",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-5-4",
        "familyName": "GPT-5.4",
        "variantType": "nano",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-5-4",
        "relatedModelKeys": [
          "gpt-5-4",
          "gpt-5-4-pro",
          "gpt-5-4-mini"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": "gpt-5-nano"
      },
      "scores": {
        "displayScore": 36,
        "overallScore": 34,
        "rawOverallScore": 34,
        "verifiedDisplayScore": 34,
        "displayCategoryScores": {
          "agentic": 16.9,
          "coding": 57,
          "reasoning": null,
          "multimodalGrounded": 15.1,
          "knowledge": 51.6,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 44.2
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 14.9,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": 15.1,
          "knowledge": 50.1,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 44.2
        }
      },
      "ranking": {
        "rankingEligible": true,
        "verifiedRankingEligible": false,
        "overallRank": 32,
        "categoryRanks": {
          "agentic": 98,
          "coding": 126,
          "knowledge": 50
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 11,
        "verifiedBenchmarkCount": 11,
        "rankableBenchmarkCount": 11,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 46.3,
          "osWorldVerified": 39,
          "mcpAtlas": 56.1,
          "toolathlon": 35.5,
          "tau2Bench": 92.5,
          "aaAgenticIndex": 29.67,
          "apexAgentsAa": 24.9,
          "gdpvalAaNormalized": 30.2,
          "gdpvalAa": 1104
        },
        "coding": {
          "vibeCodeBench": 26.097,
          "aaCodingIndex": 56.07,
          "aaSciCode": 46.9
        },
        "reasoning": {
          "lcr": 72,
          "critpt": 9.3
        },
        "multimodalGrounded": {
          "mmmuPro": 66.1,
          "mmmuProPython": 69.5,
          "aaMmmuPro": 65.4
        },
        "knowledge": {
          "gpqa": 82.8,
          "hle": 37.7,
          "hleNoTools": 24.3,
          "artificialAnalysis": 39.71,
          "aaGpqaDiamond": 81.7,
          "aaHle": 28.3,
          "aaOmniscienceIndex": -29.5,
          "omniscienceAccuracy": 25.7,
          "omniscienceHallucinationRate": 74.2
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 75.9
        },
        "math": {
          "frontierMathV2Tiers13": 25.86,
          "frontierMathV2Tier4": 6.25
        }
      }
    },
    {
      "slug": "claude-mythos-5",
      "canonicalModelKey": "claude-mythos-5",
      "model": "Claude Mythos 5",
      "creator": "Anthropic",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M+",
      "contextWindowTokens": 1000000,
      "displayScore": null,
      "provisionalDisplayScore": 86,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/claude-mythos-5",
      "markdownUrl": "https://benchlm.ai/md/models/claude-mythos-5.md",
      "id": 9,
      "releaseDate": "2026-06-09",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "claude-mythos",
        "familyName": "Claude Mythos",
        "variantType": "restricted",
        "snapshotLabel": null,
        "baseFamilyModelKey": "claude-mythos-5",
        "relatedModelKeys": [
          "claude-mythos-preview",
          "claude-mythos-5-1"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 86,
        "overallScore": 86,
        "rawOverallScore": 86,
        "verifiedDisplayScore": 86,
        "displayCategoryScores": {
          "agentic": 97.7,
          "coding": 83,
          "reasoning": null,
          "multimodalGrounded": 84.8,
          "knowledge": 97.1,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 97.7,
          "coding": 83,
          "reasoning": null,
          "multimodalGrounded": 84.8,
          "knowledge": 97.1,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": true,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": true,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 14,
        "verifiedBenchmarkCount": 14,
        "rankableBenchmarkCount": 14,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 88,
          "osWorldVerified": 85,
          "browseComp": 88,
          "cyberGym": 83.8
        },
        "coding": {
          "sweVerified": 95.5,
          "swePro": 80.3,
          "terminalBench2": 88
        },
        "reasoning": {},
        "multimodalGrounded": {
          "sweMultimodal": 54.9,
          "charxiv": 93.5,
          "charxivNoTools": 88.9
        },
        "knowledge": {
          "gpqa": 94.1,
          "hle": 64.5,
          "hleNoTools": 59
        },
        "multilingual": {
          "sweMultilingual": 92.2
        },
        "instructionFollowing": {},
        "math": {
          "usamo2026": 97.6
        },
        "external": {
          "exploitBench": 78,
          "firefox147WorkingExploit": 88.4,
          "anthropicOssFuzzAnyCrash": 80,
          "anthropicOssFuzzWritePrimitive": 32.4,
          "lastOnesCyberRangeCompletion": 60
        }
      }
    },
    {
      "slug": "sakana-fugu-ultra",
      "canonicalModelKey": "sakana-fugu-ultra",
      "model": "Sakana Fugu-Ultra",
      "creator": "Sakana AI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": null,
      "provisionalDisplayScore": 78,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/sakana-fugu-ultra",
      "markdownUrl": "https://benchlm.ai/md/models/sakana-fugu-ultra.md",
      "id": 266,
      "releaseDate": "2026-06-22",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "sakana-fugu",
        "familyName": "Sakana Fugu",
        "variantType": "ultra",
        "snapshotLabel": "Ultra",
        "baseFamilyModelKey": "sakana-fugu",
        "relatedModelKeys": [
          "sakana-fugu-ultra-v1-1",
          "sakana-fugu"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 78,
        "overallScore": 78,
        "rawOverallScore": 78,
        "verifiedDisplayScore": 78,
        "displayCategoryScores": {
          "agentic": 82.4,
          "coding": 76.8,
          "reasoning": 82.2,
          "multimodalGrounded": 71.9,
          "knowledge": 85.9,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 82.4,
          "coding": 76.8,
          "reasoning": 82.2,
          "multimodalGrounded": 71.9,
          "knowledge": 85.9,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 10,
        "verifiedBenchmarkCount": 10,
        "rankableBenchmarkCount": 10,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 82.1
        },
        "coding": {
          "swePro": 73.7,
          "terminalBench2": 82.1,
          "liveCodeBenchV6": 93.2,
          "liveCodeBenchPro": 90.8,
          "sciCode": 58.7
        },
        "reasoning": {
          "mrcrv2": 93.6
        },
        "multimodalGrounded": {
          "charxiv": 86.6
        },
        "knowledge": {
          "gpqa": 95.5,
          "gpqaDiamond": 95.5,
          "hleNoTools": 50
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "gpt-5-4-pro",
      "canonicalModelKey": "gpt-5-4-pro",
      "model": "GPT-5.4 Pro",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1.05M",
      "contextWindowTokens": 1050000,
      "displayScore": 61.47,
      "provisionalDisplayScore": 78,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 44.53,
        "upper": 78.41
      },
      "rankingEligible": true,
      "overallRank": 54,
      "url": "https://benchlm.ai/models/gpt-5-4-pro",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-5-4-pro.md",
      "id": 40,
      "releaseDate": "2026-03-05",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-5-4",
        "familyName": "GPT-5.4",
        "variantType": "pro",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-5-4",
        "relatedModelKeys": [
          "gpt-5-4",
          "gpt-5-4-mini",
          "gpt-5-4-nano"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 78,
        "overallScore": 78,
        "rawOverallScore": 77,
        "verifiedDisplayScore": 78,
        "displayCategoryScores": {
          "agentic": 81.8,
          "coding": null,
          "reasoning": 77.6,
          "multimodalGrounded": 87.6,
          "knowledge": 86.6,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 70.6
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 81.8,
          "coding": null,
          "reasoning": 77.6,
          "multimodalGrounded": 87.6,
          "knowledge": 86.6,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 70.6
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 54,
        "categoryRanks": {
          "agentic": 23
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 11,
        "verifiedBenchmarkCount": 11,
        "rankableBenchmarkCount": 11,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "browseComp": 89.3
        },
        "coding": {},
        "reasoning": {
          "arcAgi2": 83.3,
          "critpt": 30
        },
        "multimodalGrounded": {
          "mmmuPro": 94
        },
        "knowledge": {
          "hle": 58.7,
          "frontierScience": 36.7,
          "frontierScienceResearch": 36.7,
          "hleNoTools": 42.7
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {
          "ipho2025Theory": 93.5,
          "frontierMath": 50,
          "frontierMathV2Tiers13": 50,
          "frontierMathV2Tier4": 37.5
        }
      }
    },
    {
      "slug": "gpt-5-5-pro",
      "canonicalModelKey": "gpt-5-5-pro",
      "model": "GPT-5.5 Pro",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 64.35,
      "provisionalDisplayScore": 75,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 49.8,
        "upper": 78.91
      },
      "rankingEligible": true,
      "overallRank": 42,
      "url": "https://benchlm.ai/models/gpt-5-5-pro",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-5-5-pro.md",
      "id": 25,
      "releaseDate": "2026-04-23",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-5-5",
        "familyName": "GPT-5.5",
        "variantType": "pro",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-5-5",
        "relatedModelKeys": [
          "gpt-5-5"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 75,
        "overallScore": 75,
        "rawOverallScore": 75,
        "verifiedDisplayScore": 75,
        "displayCategoryScores": {
          "agentic": 83.1,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 84,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 72
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 83.1,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 84,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 72
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 42,
        "categoryRanks": {
          "agentic": 20
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 6,
        "verifiedBenchmarkCount": 6,
        "rankableBenchmarkCount": 6,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "browseComp": 90.1
        },
        "coding": {},
        "reasoning": {
          "critpt": 30.6
        },
        "multimodalGrounded": {},
        "knowledge": {
          "hle": 57.2,
          "hleNoTools": 43.1
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {
          "frontierMath": 52.4,
          "frontierMathV2Tiers13": 51,
          "frontierMathV2Tier4": 39.6
        }
      }
    },
    {
      "slug": "holo3-35b-a3b",
      "canonicalModelKey": "holo3-35b-a3b",
      "model": "Holo3-35B-A3B",
      "creator": "H Company",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "64K",
      "contextWindowTokens": 64000,
      "displayScore": null,
      "provisionalDisplayScore": 74,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/holo3-35b-a3b",
      "markdownUrl": "https://benchlm.ai/md/models/holo3-35b-a3b.md",
      "id": 3,
      "releaseDate": "2026-03-31",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "holo3",
        "familyName": "Holo3",
        "variantType": "35b-a3b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "holo3-122b-a10b",
        "relatedModelKeys": [
          "holo3-122b-a10b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": "holo2-30b-a3b"
      },
      "scores": {
        "displayScore": 74,
        "overallScore": 74,
        "rawOverallScore": 74,
        "verifiedDisplayScore": 74,
        "displayCategoryScores": {
          "agentic": 81.7,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 81.7,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {
          "agentic": 43
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "osWorldVerified": 82.56
        },
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "swe-1-7",
      "canonicalModelKey": "swe-1-7",
      "model": "SWE-1.7",
      "creator": "Cognition",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": null,
      "provisionalDisplayScore": 73,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/swe-1-7",
      "markdownUrl": "https://benchlm.ai/md/models/swe-1-7.md",
      "id": 279,
      "releaseDate": "2026-07-08",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "swe",
        "familyName": "SWE",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "swe-1-7",
        "relatedModelKeys": [
          "kimi-k2-7-code"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 73,
        "overallScore": 73,
        "rawOverallScore": 73,
        "verifiedDisplayScore": 73,
        "displayCategoryScores": {
          "agentic": 81.5,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 81.5,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {
          "agentic": 40
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 4,
        "verifiedBenchmarkCount": 4,
        "rankableBenchmarkCount": 4,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 81.5
        },
        "coding": {
          "frontierCode": 42.3,
          "terminalBench2": 81.5,
          "sweMultilingual": 77.8
        },
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "sakana-fugu",
      "canonicalModelKey": "sakana-fugu",
      "model": "Sakana Fugu",
      "creator": "Sakana AI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": null,
      "provisionalDisplayScore": 73,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/sakana-fugu",
      "markdownUrl": "https://benchlm.ai/md/models/sakana-fugu.md",
      "id": 267,
      "releaseDate": "2026-06-22",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "sakana-fugu",
        "familyName": "Sakana Fugu",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "sakana-fugu",
        "relatedModelKeys": [
          "sakana-fugu-ultra"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 73,
        "overallScore": 73,
        "rawOverallScore": 73,
        "verifiedDisplayScore": 73,
        "displayCategoryScores": {
          "agentic": 79.6,
          "coding": 66.8,
          "reasoning": 73.1,
          "multimodalGrounded": 67.8,
          "knowledge": 85.9,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 79.6,
          "coding": 66.8,
          "reasoning": 73.1,
          "multimodalGrounded": 67.8,
          "knowledge": 85.9,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 10,
        "verifiedBenchmarkCount": 10,
        "rankableBenchmarkCount": 10,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 80.2
        },
        "coding": {
          "swePro": 59,
          "terminalBench2": 80.2,
          "liveCodeBenchV6": 92.9,
          "liveCodeBenchPro": 87.8,
          "sciCode": 60.1
        },
        "reasoning": {
          "mrcrv2": 86.6
        },
        "multimodalGrounded": {
          "charxiv": 85.1
        },
        "knowledge": {
          "gpqa": 95.5,
          "gpqaDiamond": 95.5,
          "hleNoTools": 47.2
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "grok-4-6",
      "canonicalModelKey": "grok-4-6",
      "model": "Grok 4.6",
      "creator": "xAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "500K",
      "contextWindowTokens": 500000,
      "displayScore": 62.98,
      "provisionalDisplayScore": 72,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 51.47,
        "upper": 74.5
      },
      "rankingEligible": true,
      "overallRank": 47,
      "url": "https://benchlm.ai/models/grok-4-6",
      "markdownUrl": "https://benchlm.ai/md/models/grok-4-6.md",
      "id": 389,
      "releaseDate": "2026-08-12",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "grok-4-6",
        "familyName": "Grok 4.6",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "grok-4-6",
        "relatedModelKeys": [
          "grok-4-5",
          "grok-4-3",
          "grok-4-20-beta"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "grok-4-5"
      },
      "scores": {
        "displayScore": 72,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": 84.9,
          "coding": 62.1,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 68.2,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 47,
        "categoryRanks": {
          "agentic": 12,
          "coding": 24,
          "knowledge": 28
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 4,
        "verifiedBenchmarkCount": 4,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "frontierBench": 26.49,
          "apexAgents": 57.5,
          "aaAgenticIndex": 58.68,
          "gdpvalAaNormalized": 61.5,
          "gdpvalAa": 1730,
          "aaBriefcaseElo": 1551,
          "aaTau3Banking": 50.7,
          "aaTerminalBench21": 88.4,
          "aaEnterpriseOpsGym": 48.3
        },
        "coding": {
          "deepSwe": 65.9,
          "cursorBench32": 70.8,
          "frontierCode11Extended": 61.3,
          "aaCodingIndex": 76.79,
          "aaSciCode": 53.6,
          "vulcanBench": 87
        },
        "reasoning": {
          "lcr": 75,
          "critpt": 17.1
        },
        "multimodalGrounded": {
          "designArenaWebsite": 1316
        },
        "knowledge": {
          "artificialAnalysis": 60.92,
          "aaGpqaDiamond": 94.9,
          "aaHle": 42.9,
          "aaOmniscienceIndex": 30.5,
          "omniscienceAccuracy": 48.2,
          "omniscienceHallucinationRate": 34.3
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "glm-5-3",
      "canonicalModelKey": "glm-5-3",
      "model": "GLM-5.3",
      "creator": "Z.AI",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 62.41,
      "provisionalDisplayScore": 71,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 50.9,
        "upper": 73.93
      },
      "rankingEligible": true,
      "overallRank": 50,
      "url": "https://benchlm.ai/models/glm-5-3",
      "markdownUrl": "https://benchlm.ai/md/models/glm-5-3.md",
      "id": 399,
      "releaseDate": "2026-08-14",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "glm-5",
        "familyName": "GLM-5",
        "variantType": "flagship",
        "snapshotLabel": "5.3",
        "baseFamilyModelKey": "glm-5",
        "relatedModelKeys": [
          "glm-5-2",
          "glm-5-1",
          "glm-5",
          "glm-5-reasoning",
          "glm-5-turbo",
          "glm-5v-turbo"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "glm-5-2"
      },
      "scores": {
        "displayScore": 71,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": 85.1,
          "coding": 59.2,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 63.3,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 50,
        "categoryRanks": {
          "agentic": 6,
          "coding": 13,
          "knowledge": 37
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 8,
        "verifiedBenchmarkCount": 8,
        "rankableBenchmarkCount": 2,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "terminalBench21": 88.2,
          "terminalBench3": 28.3,
          "cyberGym": 84.5,
          "exploitGym": 15,
          "toolathlonVerified": 73,
          "automationBench": 48.2,
          "agentsLastExam": 28.5,
          "hleWithTools": 62.5,
          "gdpvalAa": 1769,
          "aaAgenticIndex": 59.1,
          "gdpvalAaNormalized": 62.9,
          "aaTau3Banking": 50.3,
          "aaTerminalBench21": 83.9,
          "aaBriefcaseElo": 1529
        },
        "coding": {
          "terminalBench21": 88.2,
          "terminalBench3": 28.3,
          "deepSwe": 66.9,
          "nl2Repo": 58,
          "programBench": 19,
          "frontierSwe": 78.1,
          "sweMarathon": 42.5,
          "postTrainBench": 39.8,
          "aaCodingIndex": 74.76,
          "aaSciCode": 56.5,
          "vulcanBench": 78.3,
          "openHarmonyBench": 60.8
        },
        "reasoning": {
          "lcr": 76.3,
          "critpt": 19.1
        },
        "multimodalGrounded": {},
        "knowledge": {
          "artificialAnalysis": 59.51,
          "aaGpqaDiamond": 91.7,
          "aaHle": 42.3,
          "aaOmniscienceIndex": 14.3,
          "omniscienceAccuracy": 33.9,
          "omniscienceHallucinationRate": 29.6
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {},
        "external": {
          "exploitBench": 54.4
        }
      }
    },
    {
      "slug": "exaone-4-0-32b",
      "canonicalModelKey": "exaone-4-0-32b",
      "model": "Exaone 4.0 32B",
      "creator": "LG AI Research",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": null,
      "provisionalDisplayScore": 70,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/exaone-4-0-32b",
      "markdownUrl": "https://benchlm.ai/md/models/exaone-4-0-32b.md",
      "id": 2,
      "releaseDate": null,
      "market": "Korea",
      "isRegional": true,
      "family": {
        "familyKey": "exaone-4",
        "familyName": "Exaone 4.0",
        "variantType": "32b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "exaone-4-0-32b",
        "relatedModelKeys": [
          "exaone-4-0-1-2b"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 70,
        "overallScore": 70,
        "rawOverallScore": 70,
        "verifiedDisplayScore": 70,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 79,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 79,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 2,
        "verifiedBenchmarkCount": 2,
        "rankableBenchmarkCount": 2,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 4.1
        },
        "coding": {
          "aaSciCode": 25.2
        },
        "reasoning": {
          "lcr": 9,
          "critpt": 0
        },
        "multimodalGrounded": {},
        "knowledge": {
          "mmluPro": 81.8,
          "artificialAnalysis": 5.73,
          "aaGpqaDiamond": 62.8,
          "aaHle": 5,
          "aaOmniscienceIndex": -62.8,
          "omniscienceAccuracy": 10.6,
          "omniscienceHallucinationRate": 82.1
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 33.5
        },
        "math": {
          "aime2025": 85.3
        },
        "korean": {}
      }
    },
    {
      "slug": "holo3-122b-a10b",
      "canonicalModelKey": "holo3-122b-a10b",
      "model": "Holo3-122B-A10B",
      "creator": "H Company",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "64K",
      "contextWindowTokens": 64000,
      "displayScore": null,
      "provisionalDisplayScore": 70,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/holo3-122b-a10b",
      "markdownUrl": "https://benchlm.ai/md/models/holo3-122b-a10b.md",
      "id": 13,
      "releaseDate": "2026-03-31",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "holo3",
        "familyName": "Holo3",
        "variantType": "122b-a10b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "holo3-122b-a10b",
        "relatedModelKeys": [
          "holo3-35b-a3b"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "holo2-235b-a22b"
      },
      "scores": {
        "displayScore": 70,
        "overallScore": 70,
        "rawOverallScore": 70,
        "verifiedDisplayScore": 70,
        "displayCategoryScores": {
          "agentic": 75.2,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 75.2,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {
          "agentic": 46
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "osWorldVerified": 78.85
        },
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "ornith-1-5-397b",
      "canonicalModelKey": "ornith-1-5-397b",
      "model": "Ornith-1.5-397B",
      "creator": "Ornith AI",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": 67.4,
      "provisionalDisplayScore": 69,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 57.53,
        "upper": 77.27
      },
      "rankingEligible": true,
      "overallRank": 25,
      "url": "https://benchlm.ai/models/ornith-1-5-397b",
      "markdownUrl": "https://benchlm.ai/md/models/ornith-1-5-397b.md",
      "id": 398,
      "releaseDate": "2026-08-18",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "ornith-1-5",
        "familyName": "Ornith 1.5",
        "variantType": "397b",
        "snapshotLabel": "397B",
        "baseFamilyModelKey": "ornith-1-5-397b",
        "relatedModelKeys": [
          "ornith-1-5-35b-a3b",
          "ornith-1-5-9b",
          "ornith-1-0-397b"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "ornith-1-0-397b"
      },
      "scores": {
        "displayScore": 69,
        "overallScore": 69,
        "rawOverallScore": 68,
        "verifiedDisplayScore": 69,
        "displayCategoryScores": {
          "agentic": 77.4,
          "coding": 68.7,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 65.3,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 77.4,
          "coding": 68.7,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 65.3,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 25,
        "categoryRanks": {
          "agentic": 17,
          "coding": 36
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 12,
        "verifiedBenchmarkCount": 12,
        "rankableBenchmarkCount": 12,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "terminalBench21": 86.1,
          "hleWithTools": 56.1,
          "mcpAtlas": 80,
          "toolathlonVerified": 71.2,
          "wideResearch": 80.8,
          "browseComp": 86.6,
          "clawEval": 81.4
        },
        "coding": {
          "terminalBench21": 86.1,
          "sweVerified": 86,
          "swePro": 65.1,
          "sweMultilingual": 79.6,
          "deepSwe": 56,
          "frontierBench": 13.5,
          "nl2Repo": 59.5
        },
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {
          "gpqa": 92.8,
          "gpqaDiamond": 92.8,
          "hle": 44.6,
          "hleNoTools": 44.6
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "dots3-note-preview",
      "canonicalModelKey": "dots3-note-preview",
      "model": "dots3-note Preview",
      "creator": "Dots Studio",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "512K",
      "contextWindowTokens": 512000,
      "displayScore": 68.69,
      "provisionalDisplayScore": 69,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 58.82,
        "upper": 78.55
      },
      "rankingEligible": true,
      "overallRank": 21,
      "url": "https://benchlm.ai/models/dots3-note-preview",
      "markdownUrl": "https://benchlm.ai/md/models/dots3-note-preview.md",
      "id": 394,
      "releaseDate": "2026-08-14",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "dots3-note",
        "familyName": "dots3-note",
        "variantType": "preview",
        "snapshotLabel": "Preview",
        "baseFamilyModelKey": "dots3-note-preview",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 69,
        "overallScore": 69,
        "rawOverallScore": 68,
        "verifiedDisplayScore": 69,
        "displayCategoryScores": {
          "agentic": 72,
          "coding": 56.7,
          "reasoning": 76,
          "multimodalGrounded": 64.1,
          "knowledge": 76,
          "multilingual": null,
          "instructionFollowing": 92.2,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 72,
          "coding": 56.7,
          "reasoning": 76,
          "multimodalGrounded": 64.1,
          "knowledge": 76,
          "multilingual": null,
          "instructionFollowing": 92.2,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 21,
        "categoryRanks": {
          "agentic": 19,
          "coding": 32,
          "instructionFollowing": 6
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 26,
        "verifiedBenchmarkCount": 26,
        "rankableBenchmarkCount": 26,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 3
      },
      "benchmarks": {
        "agentic": {
          "clawEval": 73.4,
          "terminalBench21": 75.1,
          "toolathlonVerified": 55.6,
          "skillsBench": 52.8,
          "apexAgents": 30.8,
          "browseComp": 83.3,
          "hleWithTools": 52.6,
          "deepSearchQa": 92.1,
          "wideResearch": 78.9
        },
        "coding": {
          "codeforces": 3056,
          "liveCodeBenchV6": 91.5,
          "terminalBench21": 75.1,
          "sweVerified": 78.4,
          "sweMultilingual": 75.7,
          "swePro": 61,
          "nl2Repo": 49.8
        },
        "reasoning": {
          "arcAgi2": 81.4
        },
        "multimodalGrounded": {
          "simpleVqa": 72.5,
          "mmmuPro": 79.1,
          "mathVision": 87.7,
          "zeroBench": 19,
          "charxivNoTools": 83.1,
          "gdpPdf": 60.7,
          "perceptionBench": 53.4,
          "babyVision": 50,
          "mmvu": 79.9,
          "videoMmmu": 86.8
        },
        "knowledge": {
          "hle": 52.6
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifBench": 80.4,
          "ifeval": 93.9
        },
        "math": {
          "imoAnswerBench": 90.9
        }
      }
    },
    {
      "slug": "gemini-3-6-flash",
      "canonicalModelKey": "gemini-3-6-flash",
      "model": "Gemini 3.6 Flash",
      "creator": "Google",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 75.06,
      "provisionalDisplayScore": 69,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 70.01,
        "upper": 80.1
      },
      "rankingEligible": true,
      "overallRank": 10,
      "url": "https://benchlm.ai/models/gemini-3-6-flash",
      "markdownUrl": "https://benchlm.ai/md/models/gemini-3-6-flash.md",
      "id": 287,
      "releaseDate": "2026-07-21",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gemini-3-6-flash",
        "familyName": "Gemini 3.6 Flash",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gemini-3-6-flash",
        "relatedModelKeys": [
          "gemini-3-5-flash",
          "gemini-3-flash"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "gemini-3-5-flash"
      },
      "scores": {
        "displayScore": 69,
        "overallScore": 74,
        "rawOverallScore": 74,
        "verifiedDisplayScore": 74,
        "displayCategoryScores": {
          "agentic": 82.5,
          "coding": 60.1,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 68.8,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 82.5,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 10,
        "categoryRanks": {
          "agentic": 118,
          "coding": 17,
          "knowledge": 25
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 2,
        "verifiedBenchmarkCount": 2,
        "rankableBenchmarkCount": 2,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "osWorldVerified": 83,
          "gdpvalAa": 1423,
          "aaAgenticIndex": 40.51,
          "gdpvalAaNormalized": 45.7,
          "aaAutomationBench": 51.1
        },
        "coding": {
          "deepSwe": 49,
          "cursorBench32": 53.5,
          "aaCodingIndex": 69.24,
          "aaSciCode": 52.7
        },
        "reasoning": {
          "lcr": 79,
          "critpt": 10.6
        },
        "multimodalGrounded": {
          "aaMmmuPro": 83.2,
          "designArenaWebsite": 1320
        },
        "knowledge": {
          "artificialAnalysis": 51.58,
          "aaGpqaDiamond": 92.8,
          "aaHle": 40.8,
          "aaOmniscienceIndex": 22.1,
          "omniscienceAccuracy": 50,
          "omniscienceHallucinationRate": 55.6
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "hy4-preview",
      "canonicalModelKey": "hy4-preview",
      "model": "Hy4 preview",
      "creator": "Tencent",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 78.3,
      "provisionalDisplayScore": 68,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 68.43,
        "upper": 88.17
      },
      "rankingEligible": true,
      "overallRank": 7,
      "url": "https://benchlm.ai/models/hy4-preview",
      "markdownUrl": "https://benchlm.ai/md/models/hy4-preview.md",
      "id": 410,
      "releaseDate": "2026-08-28",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "hy4",
        "familyName": "Hy4",
        "variantType": "preview",
        "snapshotLabel": "Preview",
        "baseFamilyModelKey": "hy4-preview",
        "relatedModelKeys": [
          "hy3"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "hy3"
      },
      "scores": {
        "displayScore": 68,
        "overallScore": 68,
        "rawOverallScore": 67,
        "verifiedDisplayScore": 68,
        "displayCategoryScores": {
          "agentic": null,
          "coding": 55.5,
          "reasoning": null,
          "multimodalGrounded": 84.5,
          "knowledge": 82.3,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": 55.5,
          "reasoning": null,
          "multimodalGrounded": 84.5,
          "knowledge": 82.3,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 7,
        "categoryRanks": {
          "agentic": 7,
          "coding": 11
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 16,
        "verifiedBenchmarkCount": 16,
        "rankableBenchmarkCount": 16,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "terminalBench21": 85.4,
          "cyberGym": 78.4,
          "wideResearch": 83.9,
          "draco": 77.2,
          "mcpAtlas": 83.7,
          "toolathlonVerified": 74.1,
          "apexAgents": 37.1,
          "skillsBench": 62.9,
          "jobBench": 61.7,
          "agentsLastExam": 22.8,
          "gdpvalAa": 1678,
          "automationBench": 32.1,
          "bankerToolBench": 78.6,
          "hleWithTools": 55.4
        },
        "coding": {
          "terminalBench21": 85.4,
          "swePro": 65.7,
          "sweMultilingual": 82.9,
          "deepSwe": 64.3,
          "nl2Repo": 58.9,
          "programBench": 17.5,
          "postTrainBench": 35.6,
          "sweMarathon": 31.9
        },
        "reasoning": {
          "critpt": 16.9
        },
        "multimodalGrounded": {
          "officeQaPro": 66.2
        },
        "knowledge": {
          "gpqa": 92.3,
          "gpqaDiamond": 92.3,
          "hle": 55.4,
          "hleNoTools": 43.4
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {
          "apex": 74.2
        }
      }
    },
    {
      "slug": "ornith-1-0-397b",
      "canonicalModelKey": "ornith-1-0-397b",
      "model": "Ornith-1.0-397B",
      "creator": "DeepReinforce AI",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": null,
      "provisionalDisplayScore": 68,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/ornith-1-0-397b",
      "markdownUrl": "https://benchlm.ai/md/models/ornith-1-0-397b.md",
      "id": 268,
      "releaseDate": "2026-06-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "ornith-1-0",
        "familyName": "Ornith 1.0",
        "variantType": "397b",
        "snapshotLabel": "397B",
        "baseFamilyModelKey": "ornith-1-0-397b",
        "relatedModelKeys": [
          "ornith-1-0-35b",
          "ornith-1-0-9b"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 68,
        "overallScore": 68,
        "rawOverallScore": 66,
        "verifiedDisplayScore": 68,
        "displayCategoryScores": {
          "agentic": 75.6,
          "coding": 62.2,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 75.6,
          "coding": 62.2,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {
          "agentic": 45,
          "coding": 49
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 6,
        "verifiedBenchmarkCount": 6,
        "rankableBenchmarkCount": 6,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 77.5,
          "clawEval": 77.1
        },
        "coding": {
          "sweVerified": 82.4,
          "swePro": 62.2,
          "sweMultilingual": 78.9,
          "nl2Repo": 48.2,
          "terminalBench2": 77.5
        },
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "glm-5-3-flash",
      "canonicalModelKey": "glm-5-3-flash",
      "model": "GLM-5.3-Flash",
      "creator": "Z.AI",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 61.25,
      "provisionalDisplayScore": 67,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 49.73,
        "upper": 72.76
      },
      "rankingEligible": true,
      "overallRank": 56,
      "url": "https://benchlm.ai/models/glm-5-3-flash",
      "markdownUrl": "https://benchlm.ai/md/models/glm-5-3-flash.md",
      "id": 408,
      "releaseDate": "2026-08-26",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "glm-5",
        "familyName": "GLM-5",
        "variantType": "flash",
        "snapshotLabel": "5.3 Flash",
        "baseFamilyModelKey": "glm-5",
        "relatedModelKeys": [
          "glm-5-3",
          "glm-5-2",
          "glm-5-1",
          "glm-5",
          "glm-5-reasoning",
          "glm-5-turbo",
          "glm-5v-turbo",
          "glm-4-7-flash"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": "glm-4-7-flash"
      },
      "scores": {
        "displayScore": 67,
        "overallScore": 72,
        "rawOverallScore": 70,
        "verifiedDisplayScore": 72,
        "displayCategoryScores": {
          "agentic": 75.6,
          "coding": 60.7,
          "reasoning": null,
          "multimodalGrounded": 80.7,
          "knowledge": 62.7,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": 80.7,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 56,
        "categoryRanks": {
          "agentic": 38,
          "coding": 20,
          "multimodalGrounded": 8,
          "knowledge": 38
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 6,
        "verifiedBenchmarkCount": 6,
        "rankableBenchmarkCount": 6,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "terminalBench21": 84.3,
          "toolathlonVerified": 78.4,
          "automationBench": 48.8,
          "agentsLastExam": 26.3,
          "hleWithTools": 55.3,
          "gdpvalAa": 1773,
          "aaTau3Banking": 47.2,
          "aaTerminalBench21": 84.3
        },
        "coding": {
          "terminalBench21": 84.3,
          "deepSwe": 63.4,
          "nl2Repo": 56.3,
          "aaSciCode": 46.1
        },
        "reasoning": {
          "lcr": 78,
          "critpt": 15.4
        },
        "multimodalGrounded": {
          "officeQaPro": 62.4,
          "charxiv": 89.4,
          "chartographyWithTools": 78,
          "babyVision": 53.4,
          "mmvu": 80.5
        },
        "knowledge": {
          "artificialAnalysis": 57.46,
          "aaGpqaDiamond": 91.2,
          "aaHle": 39.9,
          "aaOmniscienceIndex": 7.5
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "apodex-1-1",
      "canonicalModelKey": "apodex-1-1",
      "model": "Apodex 1.1",
      "creator": "Apodex",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": 56.11,
      "provisionalDisplayScore": 66,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 44.6,
        "upper": 67.63
      },
      "rankingEligible": true,
      "overallRank": 99,
      "url": "https://benchlm.ai/models/apodex-1-1",
      "markdownUrl": "https://benchlm.ai/md/models/apodex-1-1.md",
      "id": 406,
      "releaseDate": "2026-08-24",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "apodex-1-1",
        "familyName": "Apodex 1.1",
        "variantType": "full-397b",
        "snapshotLabel": "397B",
        "baseFamilyModelKey": "apodex-1-1",
        "relatedModelKeys": [
          "apodex-1-1-mini"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 66,
        "overallScore": 66,
        "rawOverallScore": 63,
        "verifiedDisplayScore": 66,
        "displayCategoryScores": {
          "agentic": null,
          "coding": 60.7,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 82,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": 60.7,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 82,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 99,
        "categoryRanks": {
          "agentic": 49,
          "coding": 45
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 8,
        "verifiedBenchmarkCount": 8,
        "rankableBenchmarkCount": 8,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "terminalBench21": 70.8,
          "hleWithTools": 56.1,
          "apexAgents": 38.5,
          "deepSearchQa": 92.4,
          "aaAgenticIndex": 36.64,
          "apexAgentsAa": 2.9,
          "gdpvalAaNormalized": 42.3,
          "gdpvalAa": 1347
        },
        "coding": {
          "terminalBench21": 70.8,
          "sweVerified": 77.7,
          "aaCodingIndex": 60.76,
          "aaSciCode": 42.9
        },
        "reasoning": {
          "lcr": 74.7,
          "critpt": 4.6
        },
        "multimodalGrounded": {
          "aaMmmuPro": 79.2
        },
        "knowledge": {
          "hle": 56.1,
          "frontierScienceResearch": 63.3,
          "bioMysteryBenchHumanDifficult": 35.3,
          "artificialAnalysis": 44,
          "aaGpqaDiamond": 86.4,
          "aaHle": 34.1,
          "aaOmniscienceIndex": -21.9,
          "omniscienceAccuracy": 31.7,
          "omniscienceHallucinationRate": 78.4
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {
          "imo2026": 30.5
        }
      }
    },
    {
      "slug": "qwen3-8-27b",
      "canonicalModelKey": "qwen3-8-27b",
      "model": "Qwen3.8-27B",
      "creator": "Alibaba",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": 71.9,
      "provisionalDisplayScore": 66,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 69.73,
        "upper": 74.07
      },
      "rankingEligible": true,
      "overallRank": 17,
      "url": "https://benchlm.ai/models/qwen3-8-27b",
      "markdownUrl": "https://benchlm.ai/md/models/qwen3-8-27b.md",
      "id": 393,
      "releaseDate": "2026-08-05",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "qwen3-8-27b",
        "familyName": "Qwen3.8-27B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "qwen3-8-27b",
        "relatedModelKeys": [
          "qwen3-8-max",
          "qwen3-6-27b",
          "qwen3-5-27b"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "qwen3-6-27b"
      },
      "scores": {
        "displayScore": 66,
        "overallScore": 66,
        "rawOverallScore": 65,
        "verifiedDisplayScore": 66,
        "displayCategoryScores": {
          "agentic": 84.8,
          "coding": 50.2,
          "reasoning": null,
          "multimodalGrounded": 81.8,
          "knowledge": 44.8,
          "multilingual": null,
          "instructionFollowing": 88.1,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 84.8,
          "coding": 48.2,
          "reasoning": null,
          "multimodalGrounded": 81.8,
          "knowledge": 42.8,
          "multilingual": null,
          "instructionFollowing": 88.1,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 17,
        "categoryRanks": {
          "agentic": 9,
          "coding": 18,
          "multimodalGrounded": 7,
          "knowledge": 55,
          "instructionFollowing": 16
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 21,
        "verifiedBenchmarkCount": 21,
        "rankableBenchmarkCount": 21,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 3
      },
      "benchmarks": {
        "agentic": {
          "terminalBench21": 73,
          "coworkBench": 70.7,
          "jobBench": 33.4,
          "agentsLastExam": 42.9,
          "osWorldVerified": 84.3,
          "webArenaVerified": 64.8,
          "androidWorld": 81.9,
          "aaAgenticIndex": 50.88,
          "gdpvalAaNormalized": 52.1,
          "gdpvalAa": 1543,
          "aaTau3Banking": 48,
          "aaTerminalBench21": 79.8,
          "aaEnterpriseOpsGym": 44.2,
          "aaBriefcaseElo": 1413
        },
        "coding": {
          "terminalBench21": 73,
          "swePro": 61.7,
          "nl2Repo": 42.3,
          "deepSwe": 42.2,
          "liveCodeBenchV6": 90.3,
          "aaCodingIndex": 68.08,
          "aaSciCode": 44.7,
          "vulcanBench": 82.6
        },
        "reasoning": {
          "lcr": 77.3,
          "critpt": 5.4
        },
        "multimodalGrounded": {
          "mathVision": 90,
          "mathVisionPython": 94.6,
          "babyVision": 65.7,
          "babyVisionPython": 85.6,
          "vision2Web": 62.9,
          "charxivNoTools": 83.7,
          "charxiv": 90.2,
          "omniDocBench15": 91.1,
          "realWorldQa": 85.9,
          "erqa": 65.5,
          "aaMmmuPro": 76.3
        },
        "knowledge": {
          "gpqa": 89.2,
          "gpqaDiamond": 89.2,
          "hle": 30.8,
          "hleNoTools": 30.8,
          "artificialAnalysis": 52.02,
          "aaGpqaDiamond": 90.5,
          "aaHle": 33.9,
          "aaOmniscienceIndex": -10,
          "omniscienceAccuracy": 15.6,
          "omniscienceHallucinationRate": 30.3,
          "aaOpennessIndex": 38.9
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifBench": 79.5
        },
        "math": {}
      }
    },
    {
      "slug": "grok-4-5",
      "canonicalModelKey": "grok-4-5",
      "model": "Grok 4.5",
      "creator": "xAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "500K",
      "contextWindowTokens": 500000,
      "displayScore": 74.9,
      "provisionalDisplayScore": 66,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 70.39,
        "upper": 79.41
      },
      "rankingEligible": true,
      "overallRank": 11,
      "url": "https://benchlm.ai/models/grok-4-5",
      "markdownUrl": "https://benchlm.ai/md/models/grok-4-5.md",
      "id": 280,
      "releaseDate": "2026-07-08",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "grok-4-5",
        "familyName": "Grok 4.5",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "grok-4-5",
        "relatedModelKeys": [
          "grok-4-3",
          "grok-4-20-beta",
          "grok-4-1"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "grok-4-3"
      },
      "scores": {
        "displayScore": 66,
        "overallScore": 64,
        "rawOverallScore": 64,
        "verifiedDisplayScore": 64,
        "displayCategoryScores": {
          "agentic": 84.2,
          "coding": 54.9,
          "reasoning": 52.1,
          "multimodalGrounded": null,
          "knowledge": 68.4,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 84.2,
          "coding": 53.7,
          "reasoning": 52.1,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 11,
        "categoryRanks": {
          "agentic": 16,
          "coding": 39,
          "knowledge": 26
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 8,
        "verifiedBenchmarkCount": 8,
        "rankableBenchmarkCount": 8,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "frontierBench": 15.68,
          "terminalBench2": 83.3,
          "deepSwe": 53,
          "aaAgenticIndex": 48.85,
          "gdpvalAaNormalized": 50.9,
          "gdpvalAa": 1518,
          "aaAutomationBench": 51.4
        },
        "coding": {
          "swePro": 64.7,
          "sweMultilingual": 78,
          "terminalBench2": 83.3,
          "cursorBench32": 66.7,
          "vulcanBench": 89.9,
          "aaCodingIndex": 72.45,
          "aaSciCode": 54.1
        },
        "reasoning": {
          "arcAgi2": 52.64,
          "arcAgi3": 0.3,
          "lcr": 74,
          "critpt": 15.4
        },
        "multimodalGrounded": {
          "aaMmmuPro": 80.4,
          "designArenaWebsite": 1303
        },
        "knowledge": {
          "artificialAnalysis": 55.76,
          "aaGpqaDiamond": 93.1,
          "aaHle": 42.7,
          "aaOmniscienceIndex": 25.3,
          "omniscienceAccuracy": 51.6,
          "omniscienceHallucinationRate": 54.1
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "claude-opus-4-7",
      "canonicalModelKey": "claude-opus-4-7",
      "model": "Claude Opus 4.7",
      "creator": "Anthropic",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 72.23,
      "provisionalDisplayScore": 65,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 60.3,
        "upper": 84.15
      },
      "rankingEligible": true,
      "overallRank": 15,
      "url": "https://benchlm.ai/models/claude-opus-4-7",
      "markdownUrl": "https://benchlm.ai/md/models/claude-opus-4-7.md",
      "id": 174,
      "releaseDate": "2026-04-16",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "claude-opus-4-7",
        "familyName": "Claude Opus 4.7",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "claude-opus-4-7",
        "relatedModelKeys": [
          "claude-opus-4-7-max",
          "claude-opus-4-6",
          "claude-opus-4-5"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "claude-opus-4-6"
      },
      "scores": {
        "displayScore": 65,
        "overallScore": 61,
        "rawOverallScore": 61,
        "verifiedDisplayScore": 61,
        "displayCategoryScores": {
          "agentic": 69.2,
          "coding": 62.3,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 63.7,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 61.8
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 61.8
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 15,
        "categoryRanks": {
          "agentic": 27,
          "coding": 12,
          "knowledge": 35
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 6,
        "verifiedBenchmarkCount": 6,
        "rankableBenchmarkCount": 6,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 74,
          "gertLabs": 65.59,
          "researchClawBench": 20.7,
          "osWorld2": 13.9
        },
        "coding": {
          "vibeCodeBench": 71.003,
          "reactNativeEvals": 82.8,
          "aaSciCode": 50.1,
          "frontierCode": 38.5
        },
        "reasoning": {
          "lcr": 72.3,
          "critpt": 5.1
        },
        "multimodalGrounded": {
          "aaMmmuPro": 76.4,
          "designArenaWebsite": 1313
        },
        "knowledge": {
          "artificialAnalysis": 43.86,
          "aaGpqaDiamond": 88.5,
          "aaHle": 33.3,
          "aaOmniscienceIndex": 14.8,
          "omniscienceAccuracy": 44.7,
          "omniscienceHallucinationRate": 54.1
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 43.6
        },
        "math": {
          "frontierMathV2Tiers13": 43.793,
          "frontierMathV2Tier4": 22.917
        }
      }
    },
    {
      "slug": "celeris-1",
      "canonicalModelKey": "celeris-1",
      "model": "Celeris-1",
      "creator": "Celeris",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": null,
      "provisionalDisplayScore": 63,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/celeris-1",
      "markdownUrl": "https://benchlm.ai/md/models/celeris-1.md",
      "id": 292,
      "releaseDate": "2026-07-22",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "celeris-1",
        "familyName": "Celeris-1",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "celeris-1",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 63,
        "overallScore": 63,
        "rawOverallScore": 63,
        "verifiedDisplayScore": 63,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 73.3,
          "multilingual": null,
          "instructionFollowing": 53,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 73.3,
          "multilingual": null,
          "instructionFollowing": 53,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {
          "instructionFollowing": 35
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 4,
        "verifiedBenchmarkCount": 4,
        "rankableBenchmarkCount": 4,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "aaAgenticIndex": 2.42,
          "gdpvalAaNormalized": 1.8,
          "gdpvalAa": 536
        },
        "coding": {
          "aaCodingIndex": 14.4,
          "aaSciCode": 20.7
        },
        "reasoning": {
          "lcr": 37.3,
          "critpt": 0,
          "drop": 81.4
        },
        "multimodalGrounded": {},
        "knowledge": {
          "mmluPro": 75.9,
          "artificialAnalysis": 12.35,
          "aaGpqaDiamond": 63.1,
          "aaHle": 6.8,
          "aaOmniscienceIndex": -71.6,
          "omniscienceAccuracy": 11,
          "omniscienceHallucinationRate": 92.8
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifeval": 80.8
        },
        "math": {
          "gsm8k": 93.7
        }
      }
    },
    {
      "slug": "gpt-5-3-codex",
      "canonicalModelKey": "gpt-5-3-codex",
      "model": "GPT-5.3 Codex",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "400K",
      "contextWindowTokens": 400000,
      "displayScore": 65.6,
      "provisionalDisplayScore": 63,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 61.48,
        "upper": 69.72
      },
      "rankingEligible": true,
      "overallRank": 34,
      "url": "https://benchlm.ai/models/gpt-5-3-codex",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-5-3-codex.md",
      "id": 5,
      "releaseDate": "2026-02-05",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-5-3-codex",
        "familyName": "GPT-5.3 Codex",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-5-3-codex",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 63,
        "overallScore": 63,
        "rawOverallScore": 62,
        "verifiedDisplayScore": 63,
        "displayCategoryScores": {
          "agentic": 64,
          "coding": 61.7,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 64,
          "coding": 61.7,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 34,
        "categoryRanks": {
          "agentic": 21,
          "coding": 25
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 7,
        "verifiedBenchmarkCount": 7,
        "rankableBenchmarkCount": 7,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 77.3,
          "osWorldVerified": 64.7,
          "tau2Bench": 86,
          "gertLabs": 57.47,
          "jobBench": 33.7
        },
        "coding": {
          "sweVerified": 85,
          "swePro": 56.8,
          "sweRebench": 58.2,
          "vibeCodeBench": 61.767,
          "aaSciCode": 53.2
        },
        "reasoning": {
          "lcr": 78.3,
          "critpt": 16.9
        },
        "multimodalGrounded": {
          "aaMmmuPro": 78.5,
          "designArenaWebsite": 1181
        },
        "knowledge": {
          "artificialAnalysis": 45.51,
          "aaGpqaDiamond": 91.5,
          "aaHle": 42.5,
          "aaOmniscienceIndex": 10.9,
          "omniscienceAccuracy": 52.9,
          "omniscienceHallucinationRate": 89.2
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 75.4
        },
        "math": {}
      }
    },
    {
      "slug": "deepseek-v3-2",
      "canonicalModelKey": "deepseek-v3-2",
      "model": "DeepSeek V3.2",
      "creator": "DeepSeek",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 54.64,
      "provisionalDisplayScore": 63,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 37.04,
        "upper": 72.24
      },
      "rankingEligible": true,
      "overallRank": 105,
      "url": "https://benchlm.ai/models/deepseek-v3-2",
      "markdownUrl": "https://benchlm.ai/md/models/deepseek-v3-2.md",
      "id": 84,
      "releaseDate": "2025-12-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "deepseek-v3-2",
        "familyName": "DeepSeek V3.2",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "deepseek-v3-2",
        "relatedModelKeys": [
          "deepseek-v3-2-thinking"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 63,
        "overallScore": 63,
        "rawOverallScore": 62,
        "verifiedDisplayScore": 63,
        "displayCategoryScores": {
          "agentic": null,
          "coding": 68.2,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 40.3
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": 68.2,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 40.3
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 105,
        "categoryRanks": {
          "coding": 83
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 6,
        "verifiedBenchmarkCount": 6,
        "rankableBenchmarkCount": 6,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "clawEval": 40.2,
          "vitaBench": 18.5,
          "tau2Bench": 78.9,
          "gertLabs": 29.57
        },
        "coding": {
          "sweRebench": 60.9,
          "reactNativeEvals": 71.5,
          "aaSciCode": 38.7
        },
        "reasoning": {
          "lcr": 42.7,
          "critpt": 0.9
        },
        "multimodalGrounded": {
          "designArenaWebsite": 1193
        },
        "knowledge": {
          "artificialAnalysis": 25.13,
          "aaGpqaDiamond": 75.1,
          "aaHle": 11.2,
          "aaOmniscienceIndex": -46.9,
          "omniscienceAccuracy": 24,
          "omniscienceHallucinationRate": 93.3
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 49
        },
        "math": {
          "frontierMathV2Tiers13": 22.1,
          "frontierMathV2Tier4": 2.1
        }
      }
    },
    {
      "slug": "composer-2-5",
      "canonicalModelKey": "composer-2-5",
      "model": "Composer 2.5",
      "creator": "Cursor",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": null,
      "provisionalDisplayScore": 62,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/composer-2-5",
      "markdownUrl": "https://benchlm.ai/md/models/composer-2-5.md",
      "id": 46,
      "releaseDate": "2026-05-18",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "composer",
        "familyName": "Composer",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "composer-2-5",
        "relatedModelKeys": [
          "composer-2",
          "composer-2-fast"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "composer-2"
      },
      "scores": {
        "displayScore": 62,
        "overallScore": 62,
        "rawOverallScore": 62,
        "verifiedDisplayScore": 62,
        "displayCategoryScores": {
          "agentic": 63.4,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 63.4,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {
          "agentic": 48
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 5,
        "verifiedBenchmarkCount": 5,
        "rankableBenchmarkCount": 5,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 69.3
        },
        "coding": {
          "terminalBench2": 69.3,
          "sweMultilingual": 79.8,
          "cursorBench31": 63.2,
          "cursorBench32": 56.1
        },
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "btl-4",
      "canonicalModelKey": "btl-4",
      "model": "BTL-4",
      "creator": "Bad Theory Labs",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": null,
      "provisionalDisplayScore": 62,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/btl-4",
      "markdownUrl": "https://benchlm.ai/md/models/btl-4.md",
      "id": 388,
      "releaseDate": "2026-08-05",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "btl",
        "familyName": "BTL",
        "variantType": "4",
        "snapshotLabel": "v4",
        "baseFamilyModelKey": "btl-4",
        "relatedModelKeys": [
          "ornith-1-0-35b"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 62,
        "overallScore": 62,
        "rawOverallScore": 58,
        "verifiedDisplayScore": 62,
        "displayCategoryScores": {
          "agentic": null,
          "coding": 61.9,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": 61.9,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {
          "agentic": 51,
          "coding": 58
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 2,
        "verifiedBenchmarkCount": 2,
        "rankableBenchmarkCount": 2,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "bfclV4": 73.5
        },
        "coding": {
          "sweVerified": 78.4,
          "liveCodeBenchV6": 66.1
        },
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "qwen3-8-flash-next",
      "canonicalModelKey": "qwen3-8-flash-next",
      "model": "Qwen3.8-Flash-Next",
      "creator": "Alibaba",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": 60.91,
      "provisionalDisplayScore": 61,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 49.39,
        "upper": 72.42
      },
      "rankingEligible": true,
      "overallRank": 64,
      "url": "https://benchlm.ai/models/qwen3-8-flash-next",
      "markdownUrl": "https://benchlm.ai/md/models/qwen3-8-flash-next.md",
      "id": 409,
      "releaseDate": "2026-08-26",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "qwen3-8-flash-next",
        "familyName": "Qwen3.8-Flash-Next",
        "variantType": "experimental-preview",
        "snapshotLabel": "Next",
        "baseFamilyModelKey": "qwen3-8-flash-next",
        "relatedModelKeys": [
          "qwen3-8-27b",
          "qwen3-8-max",
          "qwen3-7-flash",
          "qwen3-5-flash"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 61,
        "overallScore": 61,
        "rawOverallScore": 60,
        "verifiedDisplayScore": 61,
        "displayCategoryScores": {
          "agentic": null,
          "coding": 49.7,
          "reasoning": null,
          "multimodalGrounded": 82.9,
          "knowledge": 51.3,
          "multilingual": null,
          "instructionFollowing": 91.2,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": 49.7,
          "reasoning": null,
          "multimodalGrounded": 82.9,
          "knowledge": 51.3,
          "multilingual": null,
          "instructionFollowing": 91.2,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 64,
        "categoryRanks": {
          "agentic": 8,
          "coding": 26,
          "instructionFollowing": 10
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 17,
        "verifiedBenchmarkCount": 17,
        "rankableBenchmarkCount": 17,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "coworkBench": 73.9,
          "jobBench": 55.7,
          "agentsLastExam": 51.2,
          "toolathlonVerified": 73.5,
          "androidWorld": 84.5,
          "osWorld2": 19.4,
          "aaAgenticIndex": 56.42,
          "gdpvalAaNormalized": 62.1,
          "gdpvalAa": 1743
        },
        "coding": {
          "swePro": 62.5,
          "sweMultilingual": 81,
          "nl2Repo": 48.1,
          "deepSwe": 58.7,
          "liveCodeBenchV6": 91.9,
          "aaCodingIndex": 73.05,
          "aaSciCode": 46.9
        },
        "reasoning": {
          "lcr": 77,
          "critpt": 11.1
        },
        "multimodalGrounded": {
          "vision2Web": 64,
          "erqa": 72.3,
          "lvBench": 76.6,
          "realWorldQa": 88.5,
          "mathVision": 90.6,
          "mathVisionPython": 95.7,
          "charxivNoTools": 84.6,
          "charxiv": 90.6,
          "aaMmmuPro": 79.8
        },
        "knowledge": {
          "gpqa": 91.7,
          "gpqaDiamond": 91.7,
          "hle": 35.9,
          "hleNoTools": 35.9,
          "artificialAnalysis": 55.81,
          "aaGpqaDiamond": 92.3,
          "aaHle": 38,
          "aaOmniscienceIndex": -9.7,
          "omniscienceAccuracy": 24.5,
          "omniscienceHallucinationRate": 45.3
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifBench": 81.3
        },
        "math": {}
      }
    },
    {
      "slug": "o1",
      "canonicalModelKey": "o1",
      "model": "o1",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": 48.18,
      "provisionalDisplayScore": 61,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 36.66,
        "upper": 59.69
      },
      "rankingEligible": true,
      "overallRank": 146,
      "url": "https://benchlm.ai/models/o1",
      "markdownUrl": "https://benchlm.ai/md/models/o1.md",
      "id": 58,
      "releaseDate": "2024-12-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "o1",
        "familyName": "o1",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "o1",
        "relatedModelKeys": [
          "o1-pro",
          "o1-preview"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "o1-preview"
      },
      "scores": {
        "displayScore": 61,
        "overallScore": 61,
        "rawOverallScore": 61,
        "verifiedDisplayScore": 61,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 66.9,
          "multilingual": null,
          "instructionFollowing": 86.3,
          "math": 32.7
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 66.9,
          "multilingual": null,
          "instructionFollowing": 86.3,
          "math": 32.7
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 146,
        "categoryRanks": {
          "coding": 81,
          "instructionFollowing": 21
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 4,
        "verifiedBenchmarkCount": 4,
        "rankableBenchmarkCount": 4,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 62.6
        },
        "coding": {
          "aaCodingIndex": 39.72,
          "aaSciCode": 35.8
        },
        "reasoning": {
          "lcr": 63.3,
          "critpt": 0.3
        },
        "multimodalGrounded": {},
        "knowledge": {
          "mmlu": 91.8,
          "gpqa": 75.7,
          "artificialAnalysis": 23.86,
          "aaGpqaDiamond": 74.7,
          "aaHle": 7,
          "aaOmniscienceIndex": -11,
          "omniscienceAccuracy": 34.5,
          "omniscienceHallucinationRate": 69.6
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifeval": 92.2,
          "aaIfBench": 70.3
        },
        "math": {
          "frontierMathV2Tiers13": 9.31
        },
        "korean": {}
      }
    },
    {
      "slug": "mistral-medium-3-5-128b",
      "canonicalModelKey": "mistral-medium-3-5-128b",
      "model": "Mistral Medium 3.5 128B",
      "creator": "Mistral",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": null,
      "provisionalDisplayScore": 61,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/mistral-medium-3-5-128b",
      "markdownUrl": "https://benchlm.ai/md/models/mistral-medium-3-5-128b.md",
      "id": 17,
      "releaseDate": "2026-04-29",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "mistral-medium-3-5",
        "familyName": "Mistral Medium 3.5",
        "variantType": "128b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "mistral-medium-3-5-128b",
        "relatedModelKeys": [
          "mistral-medium-3"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 61,
        "overallScore": 61,
        "rawOverallScore": 57,
        "verifiedDisplayScore": 61,
        "displayCategoryScores": {
          "agentic": 55.6,
          "coding": 60.5,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": 60.5,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {
          "agentic": 83,
          "coding": 113
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 2,
        "verifiedBenchmarkCount": 2,
        "rankableBenchmarkCount": 2,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau3Bench": 91.4,
          "aaAgenticIndex": 19.21,
          "tau2Bench": 94.2,
          "gdpvalAaNormalized": 21.8,
          "gdpvalAa": 936,
          "gertLabs": 39.1,
          "aaEnterpriseOpsGym": 33.7,
          "aaHarveyLab": 69.1,
          "terminalBenchHard": 33.3,
          "aaAutomationBench": 13.7
        },
        "coding": {
          "sweVerified": 77.6,
          "aaCodingIndex": 46.9,
          "aaSciCode": 39.6
        },
        "reasoning": {
          "lcr": 65.3,
          "critpt": 0
        },
        "multimodalGrounded": {
          "aaMmmuPro": 64.9
        },
        "knowledge": {
          "artificialAnalysis": 30.39,
          "aaGpqaDiamond": 74.8,
          "aaHle": 13.8,
          "aaOmniscienceIndex": -36.8,
          "omniscienceAccuracy": 24.7,
          "omniscienceHallucinationRate": 81.6,
          "aaOpennessIndex": 33.3
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 68.8
        },
        "math": {}
      }
    },
    {
      "slug": "zaya1-8b",
      "canonicalModelKey": "zaya1-8b",
      "model": "ZAYA1-8B",
      "creator": "Zyphra",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "131K",
      "contextWindowTokens": 131000,
      "displayScore": 31.16,
      "provisionalDisplayScore": 60,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 21.29,
        "upper": 41.03
      },
      "rankingEligible": true,
      "overallRank": 210,
      "url": "https://benchlm.ai/models/zaya1-8b",
      "markdownUrl": "https://benchlm.ai/md/models/zaya1-8b.md",
      "id": 30,
      "releaseDate": "2026-05-05",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "zaya1",
        "familyName": "ZAYA1",
        "variantType": "8b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "zaya1-8b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 60,
        "overallScore": 60,
        "rawOverallScore": 59,
        "verifiedDisplayScore": 60,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 68.3,
          "multilingual": null,
          "instructionFollowing": 40.8,
          "math": 62.7
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 68.3,
          "multilingual": null,
          "instructionFollowing": 40.8,
          "math": 62.7
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 210,
        "categoryRanks": {
          "agentic": 113,
          "instructionFollowing": 38
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 10,
        "verifiedBenchmarkCount": 10,
        "rankableBenchmarkCount": 10,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "bfclV4": 39.22
        },
        "coding": {
          "liveCodeBenchV6": 65.8
        },
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {
          "gpqa": 71,
          "gpqaDiamond": 71,
          "mmluPro": 74.2
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifeval": 85.58,
          "ifBench": 52.56
        },
        "math": {
          "aime2026": 89.1,
          "hmmtFeb2026": 71.6,
          "imoAnswerBench": 59.3,
          "apex": 32.2
        }
      }
    },
    {
      "slug": "composer-2",
      "canonicalModelKey": "composer-2",
      "model": "Composer 2",
      "creator": "Cursor",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": null,
      "provisionalDisplayScore": 58,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/composer-2",
      "markdownUrl": "https://benchlm.ai/md/models/composer-2.md",
      "id": 89,
      "releaseDate": "2026-03-19",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "composer",
        "familyName": "Composer",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "composer-2",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 58,
        "overallScore": 58,
        "rawOverallScore": 58,
        "verifiedDisplayScore": 58,
        "displayCategoryScores": {
          "agentic": 52.1,
          "coding": 60.8,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 52.1,
          "coding": 60.8,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {
          "agentic": 54,
          "coding": 64
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 5,
        "verifiedBenchmarkCount": 5,
        "rankableBenchmarkCount": 5,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 61.7
        },
        "coding": {
          "sweMultilingual": 73.7,
          "sweRebench": 58,
          "reactNativeEvals": 96.1,
          "terminalBench2": 61.7
        },
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "claude-4-1-opus",
      "canonicalModelKey": "claude-4-1-opus",
      "model": "Claude 4.1 Opus",
      "creator": "Anthropic",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": 45.05,
      "provisionalDisplayScore": 58,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 42.05,
        "upper": 48.06
      },
      "rankingEligible": true,
      "overallRank": 170,
      "url": "https://benchlm.ai/models/claude-4-1-opus",
      "markdownUrl": "https://benchlm.ai/md/models/claude-4-1-opus.md",
      "id": 69,
      "releaseDate": "2025-08-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "claude-4-1-opus",
        "familyName": "Claude 4.1 Opus",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "claude-4-1-opus",
        "relatedModelKeys": [
          "claude-4-1-opus-thinking"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 58,
        "overallScore": 58,
        "rawOverallScore": 55,
        "verifiedDisplayScore": 58,
        "displayCategoryScores": {
          "agentic": null,
          "coding": 55.2,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": 55.2,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 170,
        "categoryRanks": {
          "agentic": 87,
          "coding": 107
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 2,
        "verifiedBenchmarkCount": 2,
        "rankableBenchmarkCount": 2,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "jobBench": 21.9
        },
        "coding": {
          "sweVerified": 74.5
        },
        "reasoning": {},
        "multimodalGrounded": {
          "designArenaWebsite": 1195
        },
        "knowledge": {
          "artificialAnalysis": 28.84
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "laguna-s-2-1",
      "canonicalModelKey": "laguna-s-2-1",
      "model": "Laguna S 2.1",
      "creator": "Poolside",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": null,
      "provisionalDisplayScore": 57,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/laguna-s-2-1",
      "markdownUrl": "https://benchlm.ai/md/models/laguna-s-2-1.md",
      "id": 288,
      "releaseDate": "2026-07-21",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "laguna",
        "familyName": "Laguna",
        "variantType": "s-2-1",
        "snapshotLabel": "118B-A8B",
        "baseFamilyModelKey": "laguna-s-2-1",
        "relatedModelKeys": [
          "laguna-m-1",
          "laguna-xs-2"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "laguna-m-1"
      },
      "scores": {
        "displayScore": 57,
        "overallScore": 57,
        "rawOverallScore": 57,
        "verifiedDisplayScore": 57,
        "displayCategoryScores": {
          "agentic": 64.7,
          "coding": 44,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 64.7,
          "coding": 44,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {
          "agentic": 47,
          "coding": 54
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 4,
        "verifiedBenchmarkCount": 4,
        "rankableBenchmarkCount": 4,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 70.2,
          "toolathlonVerified": 49.7
        },
        "coding": {
          "terminalBench2": 70.2,
          "sweMultilingual": 78.5,
          "swePro": 59.4,
          "deepSwe": 40.4
        },
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "qwen3-6-max-preview",
      "canonicalModelKey": "qwen3-6-max-preview",
      "model": "Qwen 3.6 Max (preview)",
      "creator": "Alibaba",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": 60.21,
      "provisionalDisplayScore": 57,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 41.15,
        "upper": 79.27
      },
      "rankingEligible": true,
      "overallRank": 69,
      "url": "https://benchlm.ai/models/qwen3-6-max-preview",
      "markdownUrl": "https://benchlm.ai/md/models/qwen3-6-max-preview.md",
      "id": 75,
      "releaseDate": "2026-04-20",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "qwen3-6-max",
        "familyName": "Qwen 3.6 Max",
        "variantType": "preview",
        "snapshotLabel": "preview",
        "baseFamilyModelKey": "qwen3-6-max-preview",
        "relatedModelKeys": [
          "qwen3-6-plus",
          "qwen3-5-plus"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 57,
        "overallScore": 57,
        "rawOverallScore": 57,
        "verifiedDisplayScore": 57,
        "displayCategoryScores": {
          "agentic": 57.6,
          "coding": 47.4,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 71,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 41.6
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 57.6,
          "coding": 47.4,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 71,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 41.6
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 69,
        "categoryRanks": {
          "agentic": 28,
          "coding": 41
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 9,
        "verifiedBenchmarkCount": 9,
        "rankableBenchmarkCount": 9,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 65.4,
          "qwenClawBench": 59,
          "qwenWebBench": 1532,
          "tau2Bench": 95.9
        },
        "coding": {
          "swePro": 57.3,
          "sciCode": 47,
          "nl2Repo": 42.9,
          "terminalBench2": 65.4,
          "aaSciCode": 46.9
        },
        "reasoning": {
          "lcr": 72,
          "critpt": 3.7
        },
        "multimodalGrounded": {},
        "knowledge": {
          "superGpqa": 73.9,
          "artificialAnalysis": 41.07,
          "aaGpqaDiamond": 88.8,
          "aaHle": 30.8,
          "aaOmniscienceIndex": 9.2,
          "omniscienceAccuracy": 37.9,
          "omniscienceHallucinationRate": 46.2
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 76.6
        },
        "math": {
          "frontierMathV2Tiers13": 23.103,
          "frontierMathV2Tier4": 4.167
        }
      }
    },
    {
      "slug": "mimo-v2-flash",
      "canonicalModelKey": "mimo-v2-flash",
      "model": "MiMo-V2-Flash",
      "creator": "Xiaomi",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": 53.25,
      "provisionalDisplayScore": 57,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 38.29,
        "upper": 68.21
      },
      "rankingEligible": true,
      "overallRank": 115,
      "url": "https://benchlm.ai/models/mimo-v2-flash",
      "markdownUrl": "https://benchlm.ai/md/models/mimo-v2-flash.md",
      "id": 42,
      "releaseDate": "2026-03-15",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "mimo-v2-flash",
        "familyName": "MiMo-V2-Flash",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "mimo-v2-flash",
        "relatedModelKeys": [
          "mimo-v2-pro",
          "mimo-v2-omni"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 57,
        "overallScore": 57,
        "rawOverallScore": 54,
        "verifiedDisplayScore": 57,
        "displayCategoryScores": {
          "agentic": null,
          "coding": 53.4,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": 53.4,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 115,
        "categoryRanks": {
          "coding": 59
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 83.9,
          "gdpvalAaNormalized": 17.3,
          "gdpvalAa": 847
        },
        "coding": {
          "sweVerified": 73.4,
          "aaCodingIndex": 49.84,
          "aaSciCode": 25.9
        },
        "reasoning": {
          "lcr": 35,
          "critpt": 0
        },
        "multimodalGrounded": {},
        "knowledge": {
          "gpqa": 83.7,
          "mmluPro": 84.9,
          "artificialAnalysis": 25.14,
          "aaGpqaDiamond": 65.6,
          "aaHle": 8.6,
          "aaOmniscienceIndex": -48.4,
          "omniscienceAccuracy": 15.6,
          "omniscienceHallucinationRate": 76
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 39.9
        },
        "math": {
          "aime2025": 94.1
        }
      }
    },
    {
      "slug": "agents-a1",
      "canonicalModelKey": "agents-a1",
      "model": "Agents-A1",
      "creator": "InternScience",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": 61.2,
      "provisionalDisplayScore": 57,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 51.33,
        "upper": 71.07
      },
      "rankingEligible": true,
      "overallRank": 58,
      "url": "https://benchlm.ai/models/agents-a1",
      "markdownUrl": "https://benchlm.ai/md/models/agents-a1.md",
      "id": 274,
      "releaseDate": "2026-06-26",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "agents-a1",
        "familyName": "Agents-A1",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "agents-a1",
        "relatedModelKeys": [
          "agents-a1-fp8",
          "agents-a1-q4-k-m-gguf",
          "agents-a1-q8-0-gguf",
          "agents-a1-f16-gguf"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 57,
        "overallScore": 57,
        "rawOverallScore": 57,
        "verifiedDisplayScore": 57,
        "displayCategoryScores": {
          "agentic": 59.3,
          "coding": null,
          "reasoning": 37.3,
          "multimodalGrounded": null,
          "knowledge": 67.3,
          "multilingual": null,
          "instructionFollowing": 93.9,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 59.3,
          "coding": null,
          "reasoning": 37.3,
          "multimodalGrounded": null,
          "knowledge": 67.3,
          "multilingual": null,
          "instructionFollowing": 93.9,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 58,
        "categoryRanks": {
          "agentic": 32,
          "instructionFollowing": 4
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 6,
        "verifiedBenchmarkCount": 6,
        "rankableBenchmarkCount": 6,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "browseComp": 75.51,
          "hleWithTools": 47.6,
          "vitaBench": 38.75
        },
        "coding": {},
        "reasoning": {
          "longBenchV2": 60.2
        },
        "multimodalGrounded": {},
        "knowledge": {
          "hle": 47.6
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifeval": 94.82
        },
        "math": {}
      }
    },
    {
      "slug": "deepseek-v4-flash-0731",
      "canonicalModelKey": "deepseek-v4-flash-max",
      "model": "DeepSeek V4 Flash 0731",
      "creator": "DeepSeek",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": null,
      "provisionalDisplayScore": 57,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/deepseek-v4-flash-0731",
      "markdownUrl": "https://benchlm.ai/md/models/deepseek-v4-flash-0731.md",
      "id": 64,
      "releaseDate": "2026-07-31",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "deepseek-v4-flash",
        "familyName": "DeepSeek V4 Flash",
        "variantType": "flash-reasoning",
        "snapshotLabel": "0731",
        "baseFamilyModelKey": "deepseek-v4-flash-max",
        "relatedModelKeys": [
          "deepseek-v4-flash-base",
          "deepseek-v4-flash",
          "deepseek-v4-flash-high"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "deepseek-v4-flash"
      },
      "scores": {
        "displayScore": 57,
        "overallScore": 55,
        "rawOverallScore": 54,
        "verifiedDisplayScore": 55,
        "displayCategoryScores": {
          "agentic": 49,
          "coding": 52.6,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 60.2,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 80.1
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 47,
          "coding": 50.6,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 59.5,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 80.1
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {
          "knowledge": 43
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 24,
        "verifiedBenchmarkCount": 24,
        "rankableBenchmarkCount": 24,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 56.9,
          "browseComp": 73.2,
          "hleWithTools": 45.1,
          "mcpAtlas": 69,
          "gdpvalAa": 1189,
          "toolathlon": 47.8,
          "terminalBench21": 82.7,
          "cyberGym": 76.7,
          "toolathlonVerified": 70.3,
          "agentsLastExam": 25.2,
          "automationBench": 25.1,
          "aaAgenticIndex": 48.41,
          "gdpvalAaNormalized": 52.4
        },
        "coding": {
          "liveCodeBenchPass1Cot": 91.6,
          "codeforces": 3052,
          "sweVerified": 79,
          "swePro": 52.6,
          "sweMultilingual": 73.3,
          "terminalBench2": 56.9,
          "terminalBench21": 82.7,
          "nl2Repo": 54.2,
          "deepSwe": 54.4,
          "dsBenchFullStack": 68.7,
          "dsBenchHard": 59.6,
          "aaCodingIndex": 69.06,
          "aaSciCode": 49.9,
          "vulcanBench": 88.4,
          "openHarmonyBench": 53.8
        },
        "reasoning": {
          "mrcr1m": 78.7,
          "corpusQa1m": 60.5,
          "lcr": 74.3,
          "critpt": 16.6
        },
        "multimodalGrounded": {
          "designArenaWebsite": 1226
        },
        "knowledge": {
          "mmluPro": 86.2,
          "simpleQa": 34.1,
          "chineseSimpleQa": 78.9,
          "gpqa": 88.1,
          "gpqaDiamond": 88.1,
          "hle": 34.8,
          "artificialAnalysis": 51.77,
          "aaGpqaDiamond": 90.8,
          "aaHle": 38.6,
          "aaOmniscienceIndex": -14.3,
          "omniscienceAccuracy": 40.4,
          "omniscienceHallucinationRate": 91.7
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {
          "hmmtFeb2026": 94.8,
          "imoAnswerBench": 88.4,
          "apex": 33,
          "apexShortlist": 85.7
        }
      }
    },
    {
      "slug": "ling-3-0-flash-fp8",
      "canonicalModelKey": "ling-3-0-flash-fp8",
      "model": "Ling 3.0 Flash FP8",
      "creator": "InclusionAI",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": null,
      "provisionalDisplayScore": 57,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/ling-3-0-flash-fp8",
      "markdownUrl": "https://benchlm.ai/md/models/ling-3-0-flash-fp8.md",
      "id": 299,
      "releaseDate": "2026-08-04",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "ling-3-0",
        "familyName": "Ling 3.0",
        "variantType": "flash-fp8",
        "snapshotLabel": "FP8",
        "baseFamilyModelKey": "ling-3-0-flash",
        "relatedModelKeys": [
          "ling-3-0-flash",
          "ling-3-0-flash-fin"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 57,
        "overallScore": 57,
        "rawOverallScore": 57,
        "verifiedDisplayScore": 57,
        "displayCategoryScores": {
          "agentic": null,
          "coding": 38.7,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 75.2,
          "multilingual": null,
          "instructionFollowing": 77.3,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": 38.7,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 75.2,
          "multilingual": null,
          "instructionFollowing": 77.3,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {
          "coding": 69,
          "instructionFollowing": 27
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 4,
        "verifiedBenchmarkCount": 4,
        "rankableBenchmarkCount": 4,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "aaAgenticIndex": 29.31,
          "gdpvalAaNormalized": 30.3,
          "gdpvalAa": 1106
        },
        "coding": {
          "sciCode": 40.37,
          "aaCodingIndex": 50.65,
          "aaSciCode": 41.1
        },
        "reasoning": {
          "lcr": 67,
          "critpt": 1.7
        },
        "multimodalGrounded": {},
        "knowledge": {
          "gpqa": 84,
          "gpqaDiamond": 84,
          "artificialAnalysis": 37.82,
          "aaGpqaDiamond": 85.5,
          "aaHle": 23.7,
          "aaOmniscienceIndex": -17.9,
          "omniscienceAccuracy": 18.2,
          "omniscienceHallucinationRate": 44.1
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifBench": 73.4
        },
        "math": {}
      }
    },
    {
      "slug": "mimo-v2-5-pro",
      "canonicalModelKey": "mimo-v2-5-pro",
      "model": "MiMo-V2.5-Pro",
      "creator": "Xiaomi",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 68.9,
      "provisionalDisplayScore": 57,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 59.51,
        "upper": 78.29
      },
      "rankingEligible": true,
      "overallRank": 20,
      "url": "https://benchlm.ai/models/mimo-v2-5-pro",
      "markdownUrl": "https://benchlm.ai/md/models/mimo-v2-5-pro.md",
      "id": 88,
      "releaseDate": "2026-04-22",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "mimo-v2-5",
        "familyName": "MiMo-V2.5",
        "variantType": "pro",
        "snapshotLabel": null,
        "baseFamilyModelKey": "mimo-v2-5",
        "relatedModelKeys": [
          "mimo-v2-5",
          "mimo-v2-pro"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": "mimo-v2-pro"
      },
      "scores": {
        "displayScore": 57,
        "overallScore": 57,
        "rawOverallScore": 56,
        "verifiedDisplayScore": 57,
        "displayCategoryScores": {
          "agentic": 62,
          "coding": 40,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 68,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 62,
          "coding": 40,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 68,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 20,
        "categoryRanks": {
          "agentic": 108,
          "coding": 46
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 7,
        "verifiedBenchmarkCount": 7,
        "rankableBenchmarkCount": 7,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "clawEval": 63.8,
          "gdpvalAa": 1265,
          "tau3Bench": 72.9,
          "terminalBench2": 68.4,
          "aaAgenticIndex": 29.52,
          "tau2Bench": 94.2,
          "gdpvalAaNormalized": 38,
          "apexAgentsAa": 2.4,
          "gertLabs": 62.7
        },
        "coding": {
          "swePro": 57.2,
          "terminalBench2": 68.4,
          "aaCodingIndex": 60.19,
          "aaSciCode": 50.2
        },
        "reasoning": {
          "lcr": 77.7,
          "critpt": 4
        },
        "multimodalGrounded": {
          "designArenaWebsite": 1293
        },
        "knowledge": {
          "hle": 48,
          "hleNoTools": 34,
          "artificialAnalysis": 42.88,
          "aaGpqaDiamond": 86.6,
          "aaHle": 35.7,
          "aaOmniscienceIndex": 3.3,
          "omniscienceAccuracy": 22.4,
          "omniscienceHallucinationRate": 24.7
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 79.9
        },
        "math": {}
      }
    },
    {
      "slug": "gemini-3-5-flash-lite",
      "canonicalModelKey": "gemini-3-5-flash-lite",
      "model": "Gemini 3.5 Flash-Lite",
      "creator": "Google",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 65.04,
      "provisionalDisplayScore": 57,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 52.89,
        "upper": 77.19
      },
      "rankingEligible": true,
      "overallRank": 37,
      "url": "https://benchlm.ai/models/gemini-3-5-flash-lite",
      "markdownUrl": "https://benchlm.ai/md/models/gemini-3-5-flash-lite.md",
      "id": 286,
      "releaseDate": "2026-07-21",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gemini-3-5-flash-lite",
        "familyName": "Gemini 3.5 Flash-Lite",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gemini-3-5-flash-lite",
        "relatedModelKeys": [
          "gemini-3-1-flash-lite"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "gemini-3-1-flash-lite"
      },
      "scores": {
        "displayScore": 57,
        "overallScore": 50,
        "rawOverallScore": 50,
        "verifiedDisplayScore": 50,
        "displayCategoryScores": {
          "agentic": 52.3,
          "coding": 36.5,
          "reasoning": 54.3,
          "multimodalGrounded": 53.8,
          "knowledge": 57.6,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 51.2,
          "coding": 34.5,
          "reasoning": 54.3,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 37,
        "categoryRanks": {
          "agentic": 132,
          "coding": 62,
          "multimodalGrounded": 23,
          "knowledge": 45
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 5,
        "verifiedBenchmarkCount": 5,
        "rankableBenchmarkCount": 5,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 54,
          "osWorldVerified": 74,
          "gdpvalAa": 1139,
          "aaAgenticIndex": 27.16,
          "gdpvalAaNormalized": 31.8,
          "aaTau3Banking": 17.5,
          "aaAutomationBench": 32.7,
          "aaEnterpriseOpsGym": 42.3
        },
        "coding": {
          "terminalBench2": 54,
          "swePro": 54.2,
          "aaCodingIndex": 49.32,
          "aaSciCode": 40.9
        },
        "reasoning": {
          "mrcrv2": 72.2,
          "lcr": 74.7,
          "critpt": 0
        },
        "multimodalGrounded": {
          "aaMmmuPro": 79
        },
        "knowledge": {
          "artificialAnalysis": 37.44,
          "aaGpqaDiamond": 83.8,
          "aaHle": 18.8,
          "aaOmniscienceIndex": 5.2,
          "omniscienceAccuracy": 29.5,
          "omniscienceHallucinationRate": 34.4
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "claude-4-sonnet",
      "canonicalModelKey": "claude-4-sonnet",
      "model": "Claude 4 Sonnet",
      "creator": "Anthropic",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": 42.07,
      "provisionalDisplayScore": 56,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 38.83,
        "upper": 45.31
      },
      "rankingEligible": true,
      "overallRank": 187,
      "url": "https://benchlm.ai/models/claude-4-sonnet",
      "markdownUrl": "https://benchlm.ai/md/models/claude-4-sonnet.md",
      "id": 67,
      "releaseDate": "2025-05-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "claude-4-sonnet",
        "familyName": "Claude 4 Sonnet",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "claude-4-sonnet",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 56,
        "overallScore": 56,
        "rawOverallScore": 53,
        "verifiedDisplayScore": 56,
        "displayCategoryScores": {
          "agentic": null,
          "coding": 52.2,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": 52.2,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 187,
        "categoryRanks": {
          "agentic": 102,
          "coding": 116
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 2,
        "verifiedBenchmarkCount": 2,
        "rankableBenchmarkCount": 2,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 52.3,
          "gertLabs": 39.66,
          "jobBench": 18.4
        },
        "coding": {
          "sweVerified": 72.7,
          "aaSciCode": 37.3
        },
        "reasoning": {
          "lcr": 45.7,
          "critpt": 1.1
        },
        "multimodalGrounded": {
          "aaMmmuPro": 62.4,
          "designArenaWebsite": 1164
        },
        "knowledge": {
          "artificialAnalysis": 25.99,
          "aaGpqaDiamond": 68.3,
          "aaHle": 4.3,
          "aaOmniscienceIndex": -9,
          "omniscienceAccuracy": 22.7,
          "omniscienceHallucinationRate": 41
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 45.4
        },
        "math": {}
      }
    },
    {
      "slug": "lfm2-5-2-6b",
      "canonicalModelKey": "lfm2-5-2-6b",
      "model": "LFM2.5-2.6B",
      "creator": "LiquidAI",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 42.9,
      "provisionalDisplayScore": 55,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 31.39,
        "upper": 54.42
      },
      "rankingEligible": true,
      "overallRank": 181,
      "url": "https://benchlm.ai/models/lfm2-5-2-6b",
      "markdownUrl": "https://benchlm.ai/md/models/lfm2-5-2-6b.md",
      "id": 300,
      "releaseDate": "2026-08-04",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "lfm2-5-2-6b",
        "familyName": "LFM2.5-2.6B",
        "variantType": "reasoning",
        "snapshotLabel": null,
        "baseFamilyModelKey": "lfm2-5-2-6b",
        "relatedModelKeys": [
          "lfm2-5-8b-a1b",
          "lfm2-5-1-2b-thinking",
          "lfm2-5-230m"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 55,
        "overallScore": 55,
        "rawOverallScore": 55,
        "verifiedDisplayScore": 55,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": 52.1,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": 52.1,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 181,
        "categoryRanks": {
          "agentic": 106,
          "coding": 123,
          "instructionFollowing": 36
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 6,
        "verifiedBenchmarkCount": 6,
        "rankableBenchmarkCount": 6,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "bfclV4": 56.88,
          "tau3Bench": 5.67,
          "clawEval": 62.85,
          "pinchBench": 68.22,
          "aaAgenticIndex": 2.41,
          "gdpvalAaNormalized": 0,
          "gdpvalAa": 254
        },
        "coding": {
          "liveCodeBenchV6": 59.41,
          "aaCodingIndex": 7.74,
          "aaSciCode": 14.2
        },
        "reasoning": {
          "lcr": 5.3,
          "critpt": 0
        },
        "multimodalGrounded": {},
        "knowledge": {
          "aaOmniscienceIndex": -29.5,
          "artificialAnalysis": 10.99,
          "aaGpqaDiamond": 55.8,
          "aaHle": 6.2,
          "omniscienceAccuracy": 4.4,
          "omniscienceHallucinationRate": 16
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifBench": 59.17
        },
        "math": {
          "aime2025": 51.87
        }
      }
    },
    {
      "slug": "kimi-k2-5-reasoning",
      "canonicalModelKey": "kimi-k2-5-reasoning",
      "model": "Kimi K2.5 (Reasoning)",
      "creator": "Moonshot AI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 60.12,
      "provisionalDisplayScore": 55,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 48.61,
        "upper": 71.63
      },
      "rankingEligible": true,
      "overallRank": 70,
      "url": "https://benchlm.ai/models/kimi-k2-5-reasoning",
      "markdownUrl": "https://benchlm.ai/md/models/kimi-k2-5-reasoning.md",
      "id": 35,
      "releaseDate": "2026-02-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "kimi-k2-5",
        "familyName": "Kimi K2.5",
        "variantType": "reasoning",
        "snapshotLabel": null,
        "baseFamilyModelKey": "kimi-k2-5",
        "relatedModelKeys": [
          "kimi-k2-5"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 55,
        "overallScore": 55,
        "rawOverallScore": 54,
        "verifiedDisplayScore": 55,
        "displayCategoryScores": {
          "agentic": 29.4,
          "coding": 59.2,
          "reasoning": null,
          "multimodalGrounded": 61.2,
          "knowledge": 83.2,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 29.4,
          "coding": 59.2,
          "reasoning": null,
          "multimodalGrounded": 61.2,
          "knowledge": 83.2,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 70,
        "categoryRanks": {
          "agentic": 63,
          "coding": 51
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 8,
        "verifiedBenchmarkCount": 8,
        "rankableBenchmarkCount": 8,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 50.8,
          "browseComp": 60.6,
          "apexAgentsAa": 11.5,
          "tau2Bench": 95.9,
          "gertLabs": 32.58,
          "aaAgenticIndex": 21.69,
          "gdpvalAaNormalized": 25.3,
          "gdpvalAa": 1006
        },
        "coding": {
          "sweVerified": 76.8,
          "vibeCodeBench": 17.536,
          "aaSciCode": 49,
          "aaCodingIndex": 46.78
        },
        "reasoning": {
          "lcr": 73,
          "critpt": 3.1
        },
        "multimodalGrounded": {
          "mmmuPro": 78.5,
          "aaMmmuPro": 75.4,
          "designArenaWebsite": 1267
        },
        "knowledge": {
          "gpqa": 87.6,
          "mmluPro": 87.1,
          "artificialAnalysis": 36.02,
          "aaGpqaDiamond": 87.9,
          "aaHle": 30.7,
          "aaOmniscienceIndex": -7.3,
          "omniscienceAccuracy": 35.2,
          "omniscienceHallucinationRate": 65.7
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 70.2
        },
        "math": {
          "aime2025": 96.1
        }
      }
    },
    {
      "slug": "gemini-3-flash",
      "canonicalModelKey": "gemini-3-flash",
      "model": "Gemini 3 Flash",
      "creator": "Google",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 59.74,
      "provisionalDisplayScore": 55,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 39.96,
        "upper": 79.52
      },
      "rankingEligible": true,
      "overallRank": 73,
      "url": "https://benchlm.ai/models/gemini-3-flash",
      "markdownUrl": "https://benchlm.ai/md/models/gemini-3-flash.md",
      "id": 77,
      "releaseDate": "2025-12-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gemini-3-flash",
        "familyName": "Gemini 3 Flash",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gemini-3-flash",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 55,
        "overallScore": 55,
        "rawOverallScore": 55,
        "verifiedDisplayScore": 55,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 50.3
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 50.3
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 73,
        "categoryRanks": {
          "agentic": 138,
          "coding": 109
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 5,
        "verifiedBenchmarkCount": 5,
        "rankableBenchmarkCount": 5,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "clawEval": 49.2,
          "tau2Bench": 43.3,
          "gertLabs": 56.63,
          "jobBench": 11.4
        },
        "coding": {
          "vibeCodeBench": 20.204,
          "aaSciCode": 49.9
        },
        "reasoning": {
          "lcr": 53,
          "critpt": 1.4
        },
        "multimodalGrounded": {
          "aaMmmuPro": 78.6,
          "designArenaWebsite": 1214
        },
        "knowledge": {
          "artificialAnalysis": 27.94,
          "aaGpqaDiamond": 81.2,
          "aaHle": 15,
          "aaOmniscienceIndex": -4.3,
          "omniscienceAccuracy": 45.8,
          "omniscienceHallucinationRate": 92.4
        },
        "multilingual": {
          "aaGlobalMmluLite": 92.7
        },
        "instructionFollowing": {
          "aaIfBench": 55.1
        },
        "math": {
          "frontierMathV2Tiers13": 35.64,
          "frontierMathV2Tier4": 4.167
        }
      }
    },
    {
      "slug": "interfaze-beta",
      "canonicalModelKey": "interfaze-beta",
      "model": "Interfaze Beta",
      "creator": "Interfaze",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": null,
      "provisionalDisplayScore": 55,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/interfaze-beta",
      "markdownUrl": "https://benchlm.ai/md/models/interfaze-beta.md",
      "id": 10,
      "releaseDate": "2026-05-11",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "interfaze",
        "familyName": "Interfaze",
        "variantType": "beta",
        "snapshotLabel": "beta",
        "baseFamilyModelKey": "interfaze-beta",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 55,
        "overallScore": 55,
        "rawOverallScore": 55,
        "verifiedDisplayScore": 55,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": 24.9,
          "knowledge": 81.2,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": 24.9,
          "knowledge": 81.2,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 10,
        "verifiedBenchmarkCount": 10,
        "rankableBenchmarkCount": 10,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {
          "spider2Lite": 52.9
        },
        "reasoning": {},
        "multimodalGrounded": {
          "ocrBenchV2": 70.7,
          "olmOcr": 85.7,
          "refcocoAvg": 82.1,
          "voxPopuliWer": 2.4,
          "mmmuPro": 71.1
        },
        "knowledge": {
          "gpqa": 89.9,
          "gpqaDiamond": 89.9,
          "mmmlu": 90.9
        },
        "multilingual": {},
        "instructionFollowing": {
          "sobValueAcc": 79.5
        },
        "math": {}
      }
    },
    {
      "slug": "muse-spark",
      "canonicalModelKey": "muse-spark",
      "model": "Muse Spark",
      "creator": "Meta",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": 70.59,
      "provisionalDisplayScore": 55,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 60.6,
        "upper": 80.58
      },
      "rankingEligible": true,
      "overallRank": 19,
      "url": "https://benchlm.ai/models/muse-spark",
      "markdownUrl": "https://benchlm.ai/md/models/muse-spark.md",
      "id": 103,
      "releaseDate": "2026-04-08",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "muse-spark",
        "familyName": "Muse Spark",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "muse-spark-1-2",
        "relatedModelKeys": [
          "muse-spark-1-2",
          "muse-spark-1-1"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 55,
        "overallScore": 55,
        "rawOverallScore": 54,
        "verifiedDisplayScore": 55,
        "displayCategoryScores": {
          "agentic": 48.1,
          "coding": 48.6,
          "reasoning": 43.7,
          "multimodalGrounded": 73.5,
          "knowledge": 72.1,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 55.9
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 48.1,
          "coding": 48.6,
          "reasoning": 43.7,
          "multimodalGrounded": 73.5,
          "knowledge": 72.1,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 55.9
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 19,
        "categoryRanks": {
          "agentic": 13,
          "coding": 22,
          "multimodalGrounded": 10
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 22,
        "verifiedBenchmarkCount": 22,
        "rankableBenchmarkCount": 22,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 3
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 59,
          "tau2Bench": 91.5,
          "deepSearchQa": 74.8,
          "cyberGym": 43.5,
          "clawEval": 63.8,
          "gdpvalAaNormalized": 32.2,
          "gdpvalAa": 1145
        },
        "coding": {
          "sweVerified": 77.4,
          "swePro": 52.4,
          "liveCodeBenchPro": 80,
          "vibeCodeBench": 19.674,
          "aaCodingIndex": 58.62,
          "aaSciCode": 51.5
        },
        "reasoning": {
          "arcAgi2": 42.5,
          "lcr": 77,
          "critpt": 11.3
        },
        "multimodalGrounded": {
          "charxiv": 86.4,
          "mmmuPro": 80.4,
          "erqa": 64.7,
          "simpleVqa": 71.3,
          "screenSpotPro": 84.1,
          "zeroBench": 33,
          "medXpertQaMm": 78.4,
          "aaMmmuPro": 80.5
        },
        "knowledge": {
          "gpqaDiamond": 89.5,
          "hle": 50.4,
          "hleNoTools": 42.8,
          "healthBenchHard": 42.8,
          "medXpertQaText": 52.6,
          "artificialAnalysis": 44.25,
          "aaGpqaDiamond": 88.4,
          "aaHle": 40.7,
          "aaOmniscienceIndex": 7.2,
          "omniscienceAccuracy": 49.6,
          "omniscienceHallucinationRate": 84.2
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 75.9
        },
        "math": {
          "frontierMathV2Tiers13": 39,
          "frontierMathV2Tier4": 14.6
        }
      }
    },
    {
      "slug": "gpt-5-1",
      "canonicalModelKey": "gpt-5-1",
      "model": "GPT-5.1",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": 53.41,
      "provisionalDisplayScore": 55,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 41.9,
        "upper": 64.93
      },
      "rankingEligible": true,
      "overallRank": 114,
      "url": "https://benchlm.ai/models/gpt-5-1",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-5-1.md",
      "id": 23,
      "releaseDate": "2025-11-13",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-5-1",
        "familyName": "GPT-5.1",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-5-1",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 55,
        "overallScore": 55,
        "rawOverallScore": 55,
        "verifiedDisplayScore": 55,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 49.7
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 49.7
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 114,
        "categoryRanks": {
          "agentic": 69,
          "coding": 60
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 3,
        "verifiedBenchmarkCount": 3,
        "rankableBenchmarkCount": 3,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "aaAgenticIndex": 21.63,
          "tau2Bench": 81.9,
          "gertLabs": 41.24,
          "gdpvalAaNormalized": 24.7,
          "gdpvalAa": 994
        },
        "coding": {
          "vibeCodeBench": 24.606,
          "aaCodingIndex": 49.39,
          "aaSciCode": 43.3
        },
        "reasoning": {
          "lcr": 76.7,
          "critpt": 4.9
        },
        "multimodalGrounded": {
          "aaMmmuPro": 75.5,
          "designArenaWebsite": 1205
        },
        "knowledge": {
          "artificialAnalysis": 37.47,
          "aaGpqaDiamond": 87.3,
          "aaHle": 28.5,
          "aaOmniscienceIndex": 5.4,
          "omniscienceAccuracy": 37.7,
          "omniscienceHallucinationRate": 51.9
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 72.9
        },
        "math": {
          "frontierMathV2Tiers13": 31.034,
          "frontierMathV2Tier4": 12.5
        },
        "korean": {}
      }
    },
    {
      "slug": "ornith-1-0-35b",
      "canonicalModelKey": "ornith-1-0-35b",
      "model": "Ornith-1.0-35B",
      "creator": "DeepReinforce AI",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": null,
      "provisionalDisplayScore": 54,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/ornith-1-0-35b",
      "markdownUrl": "https://benchlm.ai/md/models/ornith-1-0-35b.md",
      "id": 269,
      "releaseDate": "2026-06-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "ornith-1-0",
        "familyName": "Ornith 1.0",
        "variantType": "35b",
        "snapshotLabel": "35B",
        "baseFamilyModelKey": "ornith-1-0-397b",
        "relatedModelKeys": [
          "ornith-1-0-397b",
          "ornith-1-0-9b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 54,
        "overallScore": 54,
        "rawOverallScore": 53,
        "verifiedDisplayScore": 54,
        "displayCategoryScores": {
          "agentic": 55.8,
          "coding": 44.9,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 55.8,
          "coding": 44.9,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {
          "agentic": 52,
          "coding": 61
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 6,
        "verifiedBenchmarkCount": 6,
        "rankableBenchmarkCount": 6,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 64.2,
          "clawEval": 69.8
        },
        "coding": {
          "sweVerified": 75.6,
          "swePro": 50.4,
          "sweMultilingual": 69.3,
          "nl2Repo": 34.6,
          "terminalBench2": 64.2
        },
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "grok-4-3",
      "canonicalModelKey": "grok-4-3",
      "model": "Grok 4.3",
      "creator": "xAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 63.82,
      "provisionalDisplayScore": 54,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 52.9,
        "upper": 74.74
      },
      "rankingEligible": true,
      "overallRank": 44,
      "url": "https://benchlm.ai/models/grok-4-3",
      "markdownUrl": "https://benchlm.ai/md/models/grok-4-3.md",
      "id": 87,
      "releaseDate": "2026-04-30",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "grok-4-3",
        "familyName": "Grok 4.3",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "grok-4-3",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 54,
        "overallScore": 58,
        "rawOverallScore": 57,
        "verifiedDisplayScore": 58,
        "displayCategoryScores": {
          "agentic": 43.7,
          "coding": 53.3,
          "reasoning": null,
          "multimodalGrounded": 59.2,
          "knowledge": 51.6,
          "multilingual": null,
          "instructionFollowing": 91.2,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": 53,
          "reasoning": null,
          "multimodalGrounded": 59.2,
          "knowledge": 49.6,
          "multilingual": null,
          "instructionFollowing": 91.2,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 44,
        "categoryRanks": {
          "agentic": 142,
          "coding": 131,
          "knowledge": 51,
          "instructionFollowing": 11
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 6,
        "verifiedBenchmarkCount": 6,
        "rankableBenchmarkCount": 6,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 97.7,
          "gdpvalAaNormalized": 29.2,
          "aaAgenticIndex": 24.16,
          "apexAgentsAa": 17,
          "gdpvalAa": 1087,
          "gertLabs": 43.86,
          "researchClawBench": 12.4
        },
        "coding": {
          "sciCode": 47.3,
          "aaCodingIndex": 42.25,
          "aaSciCode": 47.3
        },
        "reasoning": {
          "lcr": 64.3,
          "critpt": 8
        },
        "multimodalGrounded": {
          "mmmuPro": 78.1,
          "designArenaWebsite": 1211,
          "aaMmmuPro": 78.1
        },
        "knowledge": {
          "artificialAnalysis": 37.58,
          "gpqa": 90.1,
          "hle": 35,
          "omniscienceAccuracy": 34.6,
          "omniscienceHallucinationRate": 25,
          "aaGpqaDiamond": 90.1,
          "aaHle": 37.2,
          "aaOmniscienceIndex": 18
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifBench": 81.3,
          "aaIfBench": 81.3
        },
        "math": {}
      }
    },
    {
      "slug": "claude-haiku-4-5",
      "canonicalModelKey": "claude-haiku-4-5",
      "model": "Claude Haiku 4.5",
      "creator": "Anthropic",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": 57.35,
      "provisionalDisplayScore": 54,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 45.84,
        "upper": 68.86
      },
      "rankingEligible": true,
      "overallRank": 91,
      "url": "https://benchlm.ai/models/claude-haiku-4-5",
      "markdownUrl": "https://benchlm.ai/md/models/claude-haiku-4-5.md",
      "id": 78,
      "releaseDate": "2025-10-15",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "claude-haiku-4-5",
        "familyName": "Claude Haiku 4.5",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "claude-haiku-4-5",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 54,
        "overallScore": 54,
        "rawOverallScore": 51,
        "verifiedDisplayScore": 54,
        "displayCategoryScores": {
          "agentic": null,
          "coding": 53.2,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 29
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": 53.2,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 29
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 91,
        "categoryRanks": {
          "agentic": 57,
          "coding": 95
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 5,
        "verifiedBenchmarkCount": 5,
        "rankableBenchmarkCount": 5,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "jobBench": 16
        },
        "coding": {
          "sweVerified": 73.3,
          "vulcanBench": 76.2
        },
        "reasoning": {},
        "multimodalGrounded": {
          "designArenaWebsite": 1140
        },
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {
          "frontierMathV2Tiers13": 5.903,
          "frontierMathV2Tier4": 2.083
        }
      }
    },
    {
      "slug": "mimo-v2-5",
      "canonicalModelKey": "mimo-v2-5",
      "model": "MiMo-V2.5",
      "creator": "Xiaomi",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 59.46,
      "provisionalDisplayScore": 54,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 47.94,
        "upper": 70.97
      },
      "rankingEligible": true,
      "overallRank": 78,
      "url": "https://benchlm.ai/models/mimo-v2-5",
      "markdownUrl": "https://benchlm.ai/md/models/mimo-v2-5.md",
      "id": 70,
      "releaseDate": "2026-04-22",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "mimo-v2-5",
        "familyName": "MiMo-V2.5",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "mimo-v2-5",
        "relatedModelKeys": [
          "mimo-v2-5-pro",
          "mimo-v2-omni"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "mimo-v2-omni"
      },
      "scores": {
        "displayScore": 54,
        "overallScore": 54,
        "rawOverallScore": 53,
        "verifiedDisplayScore": 54,
        "displayCategoryScores": {
          "agentic": 58.2,
          "coding": 37.9,
          "reasoning": null,
          "multimodalGrounded": 57.1,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 58.2,
          "coding": 37.9,
          "reasoning": null,
          "multimodalGrounded": 57.1,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 78,
        "categoryRanks": {
          "agentic": 44,
          "coding": 75,
          "multimodalGrounded": 22
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 8,
        "verifiedBenchmarkCount": 8,
        "rankableBenchmarkCount": 8,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "clawEval": 62.3,
          "mmClawBench": 23.8,
          "terminalBench2": 65.8,
          "gertLabs": 46.89,
          "researchClawBench": 16.9
        },
        "coding": {
          "swePro": 56.1,
          "terminalBench2": 65.8
        },
        "reasoning": {},
        "multimodalGrounded": {
          "videoMmeWithSub": 87.7,
          "charxiv": 81,
          "mmmuPro": 77.9,
          "designArenaWebsite": 1285
        },
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "qwen3-235b-2507",
      "canonicalModelKey": "qwen3-235b-2507",
      "model": "Qwen3 235B 2507",
      "creator": "Alibaba",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 56.75,
      "provisionalDisplayScore": 53,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 45.23,
        "upper": 68.26
      },
      "rankingEligible": true,
      "overallRank": 95,
      "url": "https://benchlm.ai/models/qwen3-235b-2507",
      "markdownUrl": "https://benchlm.ai/md/models/qwen3-235b-2507.md",
      "id": 144,
      "releaseDate": "2025-07-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "qwen3-235b-2507",
        "familyName": "Qwen3 235B 2507",
        "variantType": "base",
        "snapshotLabel": "2507",
        "baseFamilyModelKey": "qwen3-235b-2507",
        "relatedModelKeys": [
          "qwen3-235b-2507-reasoning"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 53,
        "overallScore": 53,
        "rawOverallScore": 52,
        "verifiedDisplayScore": 53,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 74.5,
          "multilingual": 1,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 74.5,
          "multilingual": 1,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 95,
        "categoryRanks": {
          "knowledge": 22,
          "multilingual": 12
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": true,
          "multilingual": true,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 4,
        "verifiedBenchmarkCount": 4,
        "rankableBenchmarkCount": 4,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {
          "gpqa": 77.5,
          "superGpqa": 62.6,
          "mmluPro": 83
        },
        "multilingual": {
          "mmluProX": 79.4
        },
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "gemini-3-pro-deep-think",
      "canonicalModelKey": "gemini-3-pro-deep-think",
      "model": "Gemini 3 Pro Deep Think",
      "creator": "Google",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "2M",
      "contextWindowTokens": 2000000,
      "displayScore": 62.1,
      "provisionalDisplayScore": 52,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 50.59,
        "upper": 73.62
      },
      "rankingEligible": true,
      "overallRank": 52,
      "url": "https://benchlm.ai/models/gemini-3-pro-deep-think",
      "markdownUrl": "https://benchlm.ai/md/models/gemini-3-pro-deep-think.md",
      "id": 12,
      "releaseDate": "2026-02-12",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gemini-3-pro",
        "familyName": "Gemini 3 Pro",
        "variantType": "reasoning",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gemini-3-pro",
        "relatedModelKeys": [
          "gemini-3-pro"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 52,
        "overallScore": 52,
        "rawOverallScore": 51,
        "verifiedDisplayScore": 52,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": 45.9,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": 45.9,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 52,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {
          "arcAgi2": 45.1,
          "critpt": 25.7
        },
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "o4-mini-high",
      "canonicalModelKey": "o4-mini-high",
      "model": "o4-mini (high)",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": 50.62,
      "provisionalDisplayScore": 52,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 39.11,
        "upper": 62.14
      },
      "rankingEligible": true,
      "overallRank": 132,
      "url": "https://benchlm.ai/models/o4-mini-high",
      "markdownUrl": "https://benchlm.ai/md/models/o4-mini-high.md",
      "id": 90,
      "releaseDate": "2025-04-16",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "o4-mini",
        "familyName": "o4-mini",
        "variantType": "reasoning",
        "snapshotLabel": "high",
        "baseFamilyModelKey": "o4-mini-high",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 52,
        "overallScore": 52,
        "rawOverallScore": 51,
        "verifiedDisplayScore": 52,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 43.5
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 43.5
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 132,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 2,
        "verifiedBenchmarkCount": 2,
        "rankableBenchmarkCount": 2,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {
          "frontierMathV2Tiers13": 24.828,
          "frontierMathV2Tier4": 6.25
        }
      }
    },
    {
      "slug": "mellum2-12b-a2-5b-thinking",
      "canonicalModelKey": "mellum2-12b-a2-5b-thinking",
      "model": "Mellum2-12B-A2.5B-Thinking",
      "creator": "JetBrains",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": null,
      "provisionalDisplayScore": 51,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/mellum2-12b-a2-5b-thinking",
      "markdownUrl": "https://benchlm.ai/md/models/mellum2-12b-a2-5b-thinking.md",
      "id": 252,
      "releaseDate": "2026-05-28",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "mellum2-12b-a2-5b",
        "familyName": "Mellum2 12B-A2.5B",
        "variantType": "thinking",
        "snapshotLabel": null,
        "baseFamilyModelKey": "mellum2-12b-a2-5b-instruct",
        "relatedModelKeys": [
          "mellum2-12b-a2-5b-instruct"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 51,
        "overallScore": 51,
        "rawOverallScore": 51,
        "verifiedDisplayScore": 51,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 48.7,
          "multilingual": null,
          "instructionFollowing": 40.4,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 48.7,
          "multilingual": null,
          "instructionFollowing": 40.4,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {
          "agentic": 75,
          "instructionFollowing": 39
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 5,
        "verifiedBenchmarkCount": 5,
        "rankableBenchmarkCount": 5,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "bfclV4": 45.6
        },
        "coding": {
          "liveCodeBenchV6": 69.9
        },
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {
          "mmluRedux": 86.2,
          "gpqa": 57.6,
          "gpqaDiamond": 57.6
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifeval": 76.5
        },
        "math": {}
      }
    },
    {
      "slug": "step-3-7-flash",
      "canonicalModelKey": "step-3-7-flash",
      "model": "Step 3.7 Flash",
      "creator": "StepFun",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": 50.83,
      "provisionalDisplayScore": 51,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 39.31,
        "upper": 62.34
      },
      "rankingEligible": true,
      "overallRank": 127,
      "url": "https://benchlm.ai/models/step-3-7-flash",
      "markdownUrl": "https://benchlm.ai/md/models/step-3-7-flash.md",
      "id": 239,
      "releaseDate": "2026-05-29",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "step-3-7-flash",
        "familyName": "Step 3.7 Flash",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "step-3-7-flash",
        "relatedModelKeys": [
          "step-3-5-flash"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "step-3-5-flash"
      },
      "scores": {
        "displayScore": 51,
        "overallScore": 51,
        "rawOverallScore": 50,
        "verifiedDisplayScore": 51,
        "displayCategoryScores": {
          "agentic": 52,
          "coding": 38.3,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 52,
          "coding": 38.3,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 127,
        "categoryRanks": {
          "agentic": 84,
          "coding": 74
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 10,
        "verifiedBenchmarkCount": 10,
        "rankableBenchmarkCount": 10,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 59.5,
          "browseComp": 75.82,
          "deepSearchQa": 92.82,
          "gdpvalAaNormalized": 25.8,
          "toolathlon": 49.5,
          "clawEval": 67.1,
          "hleWithTools": 47.2,
          "gertLabs": 51.57,
          "aaAgenticIndex": 21.74,
          "tau2Bench": 98.5,
          "gdpvalAa": 1018,
          "apexAgentsAa": 14.8
        },
        "coding": {
          "swePro": 56.3,
          "terminalBench2": 59.5,
          "aaCodingIndex": 39.57,
          "aaSciCode": 40
        },
        "reasoning": {
          "lcr": 69.7,
          "critpt": 2.3
        },
        "multimodalGrounded": {
          "simpleVqa": 79.2,
          "vStar": 95.3,
          "aaMmmuPro": 75.3,
          "designArenaWebsite": 1211
        },
        "knowledge": {
          "artificialAnalysis": 30.9,
          "aaGpqaDiamond": 80.9,
          "aaHle": 21.4,
          "aaOmniscienceIndex": -37.3,
          "omniscienceAccuracy": 25.8,
          "omniscienceHallucinationRate": 85
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 67.3
        },
        "math": {}
      }
    },
    {
      "slug": "ornith-1-5-35b-a3b",
      "canonicalModelKey": "ornith-1-5-35b-a3b",
      "model": "Ornith-1.5-35B-A3B",
      "creator": "Ornith AI",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": 47.9,
      "provisionalDisplayScore": 51,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 38.03,
        "upper": 57.77
      },
      "rankingEligible": true,
      "overallRank": 149,
      "url": "https://benchlm.ai/models/ornith-1-5-35b-a3b",
      "markdownUrl": "https://benchlm.ai/md/models/ornith-1-5-35b-a3b.md",
      "id": 397,
      "releaseDate": "2026-08-18",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "ornith-1-5",
        "familyName": "Ornith 1.5",
        "variantType": "35b-a3b",
        "snapshotLabel": "35B-A3B",
        "baseFamilyModelKey": "ornith-1-5-397b",
        "relatedModelKeys": [
          "ornith-1-5-397b",
          "ornith-1-5-9b",
          "ornith-1-0-35b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": "ornith-1-0-35b"
      },
      "scores": {
        "displayScore": 51,
        "overallScore": 51,
        "rawOverallScore": 49,
        "verifiedDisplayScore": 51,
        "displayCategoryScores": {
          "agentic": 46.3,
          "coding": 56.2,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 34.5,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 46.3,
          "coding": 56.2,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 34.5,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 149,
        "categoryRanks": {
          "agentic": 73,
          "coding": 82
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 12,
        "verifiedBenchmarkCount": 12,
        "rankableBenchmarkCount": 12,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "terminalBench21": 67.8,
          "hleWithTools": 33.4,
          "mcpAtlas": 70.2,
          "toolathlonVerified": 48.7,
          "wideResearch": 67.8,
          "browseComp": 67.6,
          "clawEval": 72.5
        },
        "coding": {
          "terminalBench21": 67.8,
          "sweVerified": 79,
          "swePro": 59.6,
          "sweMultilingual": 71.4,
          "deepSwe": 22,
          "frontierBench": 5.1,
          "nl2Repo": 46.2
        },
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {
          "gpqa": 89.2,
          "gpqaDiamond": 89.2,
          "hle": 25.6,
          "hleNoTools": 25.6
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "o3-mini",
      "canonicalModelKey": "o3-mini",
      "model": "o3-mini",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": 46.69,
      "provisionalDisplayScore": 50,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 31.65,
        "upper": 61.73
      },
      "rankingEligible": true,
      "overallRank": 158,
      "url": "https://benchlm.ai/models/o3-mini",
      "markdownUrl": "https://benchlm.ai/md/models/o3-mini.md",
      "id": 41,
      "releaseDate": "2025-01-31",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "o3",
        "familyName": "o3",
        "variantType": "mini",
        "snapshotLabel": null,
        "baseFamilyModelKey": "o3",
        "relatedModelKeys": [
          "o3",
          "o3-pro"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 50,
        "overallScore": 50,
        "rawOverallScore": 50,
        "verifiedDisplayScore": 50,
        "displayCategoryScores": {
          "agentic": null,
          "coding": 21.3,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 68.4,
          "multilingual": null,
          "instructionFollowing": 91.2,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": 21.3,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 68.4,
          "multilingual": null,
          "instructionFollowing": 91.2,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 158,
        "categoryRanks": {
          "coding": 96,
          "instructionFollowing": 12
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 5,
        "verifiedBenchmarkCount": 5,
        "rankableBenchmarkCount": 5,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 28.7
        },
        "coding": {
          "sweVerified": 49.3,
          "aaSciCode": 39.9
        },
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {
          "mmlu": 86.9,
          "gpqa": 77.2,
          "artificialAnalysis": 19.22,
          "aaGpqaDiamond": 74.8,
          "aaHle": 7.9
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifeval": 93.9
        },
        "math": {
          "aime2024": 87.3
        }
      }
    },
    {
      "slug": "soofi-s-30b-a3b",
      "canonicalModelKey": "soofi-s-30b-a3b",
      "model": "Soofi S 30B-A3B",
      "creator": "Soofi Project",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": null,
      "provisionalDisplayScore": 50,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/soofi-s-30b-a3b",
      "markdownUrl": "https://benchlm.ai/md/models/soofi-s-30b-a3b.md",
      "id": 283,
      "releaseDate": "2026-07-10",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "soofi-s",
        "familyName": "Soofi S",
        "variantType": "base",
        "snapshotLabel": "30B-A3B",
        "baseFamilyModelKey": "soofi-s-30b-a3b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 50,
        "overallScore": 50,
        "rawOverallScore": 50,
        "verifiedDisplayScore": 50,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 42.6,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 42.6,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 8,
        "verifiedBenchmarkCount": 8,
        "rankableBenchmarkCount": 8,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {
          "humaneval": 73.8
        },
        "reasoning": {
          "bbh": 78.8,
          "drop": 66.5
        },
        "multimodalGrounded": {},
        "knowledge": {
          "gpqa": 43.4,
          "gpqaDiamond": 43.4,
          "mmluPro": 51.4,
          "agieval": 66.9
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {
          "gsm8k": 86.1
        }
      }
    },
    {
      "slug": "lfm2-5-8b-a1b",
      "canonicalModelKey": "lfm2-5-8b-a1b",
      "model": "LFM2.5-8B-A1B",
      "creator": "LiquidAI",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 41.78,
      "provisionalDisplayScore": 50,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 30.27,
        "upper": 53.3
      },
      "rankingEligible": true,
      "overallRank": 188,
      "url": "https://benchlm.ai/models/lfm2-5-8b-a1b",
      "markdownUrl": "https://benchlm.ai/md/models/lfm2-5-8b-a1b.md",
      "id": 240,
      "releaseDate": "2026-05-28",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "lfm2-5-8b-a1b",
        "familyName": "LFM2.5-8B-A1B",
        "variantType": "reasoning",
        "snapshotLabel": null,
        "baseFamilyModelKey": "lfm2-5-8b-a1b",
        "relatedModelKeys": [
          "lfm2-5-1-2b-thinking",
          "lfm2-5-350m"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 50,
        "overallScore": 50,
        "rawOverallScore": 49,
        "verifiedDisplayScore": 50,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": 55.7,
          "math": 27.8
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": 55.7,
          "math": 27.8
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 188,
        "categoryRanks": {
          "agentic": 81,
          "instructionFollowing": 34
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 6,
        "verifiedBenchmarkCount": 6,
        "rankableBenchmarkCount": 6,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "bfclV4": 49.73,
          "tau2Bench": 88.07
        },
        "coding": {
          "aaSciCode": 7.8
        },
        "reasoning": {
          "lcr": 0,
          "critpt": 0
        },
        "multimodalGrounded": {},
        "knowledge": {
          "aaGpqaDiamond": 51.3,
          "aaHle": 6.9,
          "aaOmniscienceIndex": -33.3,
          "omniscienceAccuracy": 9.4,
          "omniscienceHallucinationRate": 46.9,
          "artificialAnalysis": 8.14
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifeval": 91.84,
          "ifBench": 56.47,
          "aaIfBench": 55.6
        },
        "math": {
          "math500": 88.76,
          "aime2025": 42.53,
          "aime2026": 50
        }
      }
    },
    {
      "slug": "qwen3-5-plus",
      "canonicalModelKey": "qwen3-5-plus",
      "model": "Qwen3.5 Plus",
      "creator": "Alibaba",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 47.79,
      "provisionalDisplayScore": 50,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 33.82,
        "upper": 61.76
      },
      "rankingEligible": true,
      "overallRank": 150,
      "url": "https://benchlm.ai/models/qwen3-5-plus",
      "markdownUrl": "https://benchlm.ai/md/models/qwen3-5-plus.md",
      "id": 227,
      "releaseDate": "2026-03-04",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "qwen3-5-plus",
        "familyName": "Qwen3.5 Plus",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "qwen3-5-plus",
        "relatedModelKeys": [
          "qwen3-5-397b",
          "qwen3-5-397b-reasoning"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 50,
        "overallScore": 50,
        "rawOverallScore": 49,
        "verifiedDisplayScore": 50,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 39.5
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 39.5
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 150,
        "categoryRanks": {
          "agentic": 66
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 4,
        "verifiedBenchmarkCount": 4,
        "rankableBenchmarkCount": 4,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "jobBench": 18.5
        },
        "coding": {
          "vibeCodeBench": 15.738
        },
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {
          "frontierMathV2Tiers13": 21.034,
          "frontierMathV2Tier4": 2.083
        }
      }
    },
    {
      "slug": "kimi-k2",
      "canonicalModelKey": "kimi-k2",
      "model": "Kimi K2",
      "creator": "Moonshot AI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 26.24,
      "provisionalDisplayScore": 49,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 16.05,
        "upper": 36.43
      },
      "rankingEligible": true,
      "overallRank": 213,
      "url": "https://benchlm.ai/models/kimi-k2",
      "markdownUrl": "https://benchlm.ai/md/models/kimi-k2.md",
      "id": 96,
      "releaseDate": "2025-07-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "kimi-k2",
        "familyName": "Kimi K2",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "kimi-k2",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 49,
        "overallScore": 49,
        "rawOverallScore": 49,
        "verifiedDisplayScore": 49,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 39.2
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 39.2
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 213,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 2,
        "verifiedBenchmarkCount": 2,
        "rankableBenchmarkCount": 2,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 61.1
        },
        "coding": {
          "aaSciCode": 34.5
        },
        "reasoning": {
          "lcr": 53,
          "critpt": 0
        },
        "multimodalGrounded": {
          "designArenaWebsite": 1068
        },
        "knowledge": {
          "artificialAnalysis": 19.65,
          "aaGpqaDiamond": 76.6,
          "aaHle": 7.4,
          "aaOmniscienceIndex": -28.3,
          "omniscienceAccuracy": 27.4,
          "omniscienceHallucinationRate": 76.6
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 41.5
        },
        "math": {
          "frontierMathV2Tiers13": 21.404,
          "frontierMathV2Tier4": 0
        }
      }
    },
    {
      "slug": "minimax-m2-7",
      "canonicalModelKey": "minimax-m2-7",
      "model": "MiniMax M2.7",
      "creator": "MiniMax",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": 62.85,
      "provisionalDisplayScore": 49,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 55,
        "upper": 70.7
      },
      "rankingEligible": true,
      "overallRank": 49,
      "url": "https://benchlm.ai/models/minimax-m2-7",
      "markdownUrl": "https://benchlm.ai/md/models/minimax-m2-7.md",
      "id": 117,
      "releaseDate": "2026-03-18",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "minimax-m2-7",
        "familyName": "MiniMax M2.7",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "minimax-m2-7",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "minimax-m2-5"
      },
      "scores": {
        "displayScore": 49,
        "overallScore": 49,
        "rawOverallScore": 49,
        "verifiedDisplayScore": 49,
        "displayCategoryScores": {
          "agentic": 45.1,
          "coding": 41.9,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 45.1,
          "coding": 41.9,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 49,
        "categoryRanks": {
          "agentic": 126,
          "coding": 110
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 16,
        "verifiedBenchmarkCount": 16,
        "rankableBenchmarkCount": 16,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 57,
          "tau2Bench": 84.8,
          "toolathlon": 46.3,
          "mleBenchLite": 66.6,
          "mmClawBench": 62.7,
          "clawEval": 48.7,
          "aaAgenticIndex": 25.92,
          "apexAgentsAa": 10.6,
          "gdpvalAaNormalized": 32.9,
          "gdpvalAa": 1157,
          "gertLabs": 40.4
        },
        "coding": {
          "sweVerifiedArcee": 75.4,
          "swePro": 56.2,
          "sweRebench": 51.9,
          "sweMultilingual": 76.5,
          "multiSweBench": 52.7,
          "vibePro": 55.6,
          "nl2Repo": 39.8,
          "vibeCodeBench": 27.037,
          "reactNativeEvals": 71.4,
          "aaCodingIndex": 52.62,
          "aaSciCode": 47
        },
        "reasoning": {
          "lcr": 75.3,
          "critpt": 0.6
        },
        "multimodalGrounded": {
          "designArenaWebsite": 1263
        },
        "knowledge": {
          "gpqaDiamond": 87,
          "mmluProArcee": 80.8,
          "artificialAnalysis": 38.87,
          "aaGpqaDiamond": 87.4,
          "aaHle": 29.6,
          "aaOmniscienceIndex": 0.8,
          "omniscienceAccuracy": 26.8,
          "omniscienceHallucinationRate": 35.6
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 75.7
        },
        "math": {
          "aime2025Arcee": 80
        }
      }
    },
    {
      "slug": "grok-4",
      "canonicalModelKey": "grok-4",
      "model": "Grok 4",
      "creator": "xAI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 59.42,
      "provisionalDisplayScore": 49,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 48.8,
        "upper": 70.04
      },
      "rankingEligible": true,
      "overallRank": 79,
      "url": "https://benchlm.ai/models/grok-4",
      "markdownUrl": "https://benchlm.ai/md/models/grok-4.md",
      "id": 56,
      "releaseDate": "2025-07-09",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "grok-4",
        "familyName": "Grok 4",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "grok-4",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 49,
        "overallScore": 49,
        "rawOverallScore": 49,
        "verifiedDisplayScore": 49,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 38.6
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 38.6
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 79,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 3,
        "verifiedBenchmarkCount": 3,
        "rankableBenchmarkCount": 3,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 74.9,
          "gertLabs": 42.34
        },
        "coding": {
          "reactNativeEvals": 72.6,
          "aaSciCode": 45.7
        },
        "reasoning": {
          "lcr": 67,
          "critpt": 2
        },
        "multimodalGrounded": {
          "aaMmmuPro": 68.8
        },
        "knowledge": {
          "artificialAnalysis": 34.08,
          "aaGpqaDiamond": 87.7,
          "aaHle": 26.7,
          "aaOmniscienceIndex": 2.1,
          "omniscienceAccuracy": 40.5,
          "omniscienceHallucinationRate": 64.5
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 53.7
        },
        "math": {
          "frontierMathV2Tiers13": 19.655,
          "frontierMathV2Tier4": 2.083
        }
      }
    },
    {
      "slug": "o3",
      "canonicalModelKey": "o3",
      "model": "o3",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": 46.82,
      "provisionalDisplayScore": 49,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 43.88,
        "upper": 49.76
      },
      "rankingEligible": true,
      "overallRank": 157,
      "url": "https://benchlm.ai/models/o3",
      "markdownUrl": "https://benchlm.ai/md/models/o3.md",
      "id": 50,
      "releaseDate": "2025-04-16",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "o3",
        "familyName": "o3",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "o3",
        "relatedModelKeys": [
          "o3-pro",
          "o3-mini"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 49,
        "overallScore": 49,
        "rawOverallScore": 48,
        "verifiedDisplayScore": 49,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 37.9
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 37.9
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 157,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 2,
        "verifiedBenchmarkCount": 2,
        "rankableBenchmarkCount": 2,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 80.7
        },
        "coding": {
          "aaSciCode": 41
        },
        "reasoning": {
          "lcr": 73.3,
          "critpt": 1.1
        },
        "multimodalGrounded": {
          "aaMmmuPro": 70.1,
          "designArenaWebsite": 1054
        },
        "knowledge": {
          "artificialAnalysis": 31.1,
          "aaGpqaDiamond": 82.7,
          "aaHle": 20.1,
          "aaOmniscienceIndex": -15.5,
          "omniscienceAccuracy": 38.6,
          "omniscienceHallucinationRate": 88.1
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 71.4
        },
        "math": {
          "aaMath500": 99.2,
          "frontierMathV2Tiers13": 18.685,
          "frontierMathV2Tier4": 2.083
        }
      }
    },
    {
      "slug": "granite-4-2-8b",
      "canonicalModelKey": "granite-4-2-8b",
      "model": "Granite 4.2 8B",
      "creator": "IBM",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 46.32,
      "provisionalDisplayScore": 48,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 34.8,
        "upper": 57.83
      },
      "rankingEligible": true,
      "overallRank": 161,
      "url": "https://benchlm.ai/models/granite-4-2-8b",
      "markdownUrl": "https://benchlm.ai/md/models/granite-4-2-8b.md",
      "id": 413,
      "releaseDate": "2026-08-25",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "granite-4-2",
        "familyName": "Granite 4.2",
        "variantType": "8b-dense",
        "snapshotLabel": "8B",
        "baseFamilyModelKey": "granite-4-2-8b",
        "relatedModelKeys": [
          "granite-4-0-h-1b",
          "granite-4-0-1b"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 48,
        "overallScore": 48,
        "rawOverallScore": 49,
        "verifiedDisplayScore": 48,
        "displayCategoryScores": {
          "agentic": null,
          "coding": 18.8,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 66.5,
          "multilingual": null,
          "instructionFollowing": 87.8,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": 18.8,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 66.5,
          "multilingual": null,
          "instructionFollowing": 87.8,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 161,
        "categoryRanks": {
          "agentic": 93,
          "coding": 105,
          "instructionFollowing": 17
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 11,
        "verifiedBenchmarkCount": 11,
        "rankableBenchmarkCount": 11,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "terminalBench21": 20.56,
          "tau3Bench": 58.06,
          "bfclV4": 52.39,
          "aaAgenticIndex": 9.17,
          "gdpvalAaNormalized": 10.1,
          "gdpvalAa": 702
        },
        "coding": {
          "sweVerified": 47.67,
          "swePro": 19.11,
          "sweMultilingual": 30.78,
          "terminalBench21": 20.56,
          "liveCodeBenchV6": 73.24,
          "sciCode": 36.09,
          "aaCodingIndex": 22.38,
          "aaSciCode": 30.4
        },
        "reasoning": {
          "lcr": 43.3,
          "critpt": 0.3
        },
        "multimodalGrounded": {},
        "knowledge": {
          "gpqa": 64.14,
          "mmluPro": 74.04,
          "artificialAnalysis": 19.61,
          "aaGpqaDiamond": 63.1,
          "aaHle": 9.7,
          "aaOmniscienceIndex": -17.2,
          "omniscienceAccuracy": 11.2,
          "omniscienceHallucinationRate": 32
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifBench": 79.33
        },
        "math": {
          "aime2025": 86.67,
          "hmmtFeb2025": 78.33
        }
      }
    },
    {
      "slug": "nemotron-3-nano-omni-30b-a3b",
      "canonicalModelKey": "nemotron-3-nano-omni-30b-a3b",
      "model": "Nemotron 3 Nano Omni 30B A3B",
      "creator": "NVIDIA",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": 44.49,
      "provisionalDisplayScore": 48,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 32.98,
        "upper": 56
      },
      "rankingEligible": true,
      "overallRank": 172,
      "url": "https://benchlm.ai/models/nemotron-3-nano-omni-30b-a3b",
      "markdownUrl": "https://benchlm.ai/md/models/nemotron-3-nano-omni-30b-a3b.md",
      "id": 52,
      "releaseDate": "2026-04-28",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "nemotron-3-nano-omni",
        "familyName": "Nemotron 3 Nano Omni",
        "variantType": "30b-a3b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "nemotron-3-nano-omni-30b-a3b",
        "relatedModelKeys": [
          "nemotron-3-nano-30b",
          "nemotron-3-super-120b-a12b",
          "nemotron-3-super-100b",
          "nemotron-3-ultra-500b"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 48,
        "overallScore": 48,
        "rawOverallScore": 48,
        "verifiedDisplayScore": 48,
        "displayCategoryScores": {
          "agentic": null,
          "coding": 21.4,
          "reasoning": null,
          "multimodalGrounded": 43.5,
          "knowledge": 71.2,
          "multilingual": null,
          "instructionFollowing": 78.7,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": 21.4,
          "reasoning": null,
          "multimodalGrounded": 43.5,
          "knowledge": 71.2,
          "multilingual": null,
          "instructionFollowing": 78.7,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 172,
        "categoryRanks": {
          "coding": 115,
          "instructionFollowing": 26
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 14,
        "verifiedBenchmarkCount": 14,
        "rankableBenchmarkCount": 14,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "osWorld": 47.4,
          "tau2Bench": 45.3,
          "gdpvalAaNormalized": 0,
          "gdpvalAa": 469
        },
        "coding": {
          "liveCodeBenchV5": 63.2,
          "sciCode": 32,
          "aaSciCode": 27.8,
          "aaCodingIndex": 13.75
        },
        "reasoning": {
          "lcr": 40.7,
          "critpt": 0
        },
        "multimodalGrounded": {
          "mmmu": 70.8,
          "mmLongBenchDoc": 57.5,
          "charxiv": 76.25,
          "screenSpotPro": 57.8,
          "videoMmeNoSub": 72.2,
          "ai2dTest": 88.5,
          "refcocoAvg": 90.5,
          "aaMmmuPro": 53.2
        },
        "knowledge": {
          "mmluPro": 77.3,
          "gpqa": 72.2,
          "gpqaDiamond": 72.2,
          "artificialAnalysis": 15.01,
          "aaGpqaDiamond": 46.9,
          "aaHle": 4.8,
          "aaOmniscienceIndex": -57.4,
          "omniscienceAccuracy": 15.2,
          "omniscienceHallucinationRate": 85.7
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifBench": 74.2,
          "aaIfBench": 63.2
        },
        "math": {
          "aime2025": 82.1
        }
      }
    },
    {
      "slug": "gpt-4-1-nano",
      "canonicalModelKey": "gpt-4-1-nano",
      "model": "GPT-4.1 nano",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 42.38,
      "provisionalDisplayScore": 48,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 30.86,
        "upper": 53.89
      },
      "rankingEligible": true,
      "overallRank": 183,
      "url": "https://benchlm.ai/models/gpt-4-1-nano",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-4-1-nano.md",
      "id": 121,
      "releaseDate": "2025-04-14",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-4-1",
        "familyName": "GPT-4.1",
        "variantType": "nano",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-4-1",
        "relatedModelKeys": [
          "gpt-4-1",
          "gpt-4-1-mini"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 48,
        "overallScore": 48,
        "rawOverallScore": 48,
        "verifiedDisplayScore": 48,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 41.4,
          "multilingual": null,
          "instructionFollowing": 60,
          "math": 25.6
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 41.4,
          "multilingual": null,
          "instructionFollowing": 60,
          "math": 25.6
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 183,
        "categoryRanks": {
          "agentic": 109,
          "coding": 118,
          "instructionFollowing": 31
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 4,
        "verifiedBenchmarkCount": 4,
        "rankableBenchmarkCount": 4,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "aaAgenticIndex": 1.17,
          "tau2Bench": 17.3,
          "gdpvalAaNormalized": 0,
          "gdpvalAa": 62
        },
        "coding": {
          "aaCodingIndex": 11.14,
          "aaSciCode": 25.9
        },
        "reasoning": {
          "lcr": 19.3,
          "critpt": 0
        },
        "multimodalGrounded": {
          "aaMmmuPro": 40.1,
          "designArenaWebsite": 991
        },
        "knowledge": {
          "mmlu": 80.1,
          "gpqa": 50.3,
          "artificialAnalysis": 9.64,
          "aaGpqaDiamond": 51.2,
          "aaHle": 3.8,
          "aaOmniscienceIndex": -57.6,
          "omniscienceAccuracy": 13.7,
          "omniscienceHallucinationRate": 82.6
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifeval": 83.2,
          "aaIfBench": 32
        },
        "math": {
          "frontierMathV2Tiers13": 1.034
        },
        "korean": {}
      }
    },
    {
      "slug": "pokee-isaac-28b",
      "canonicalModelKey": "pokee-isaac-28b",
      "model": "Pokee-Isaac 28B",
      "creator": "Pokee AI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "10M",
      "contextWindowTokens": 10000000,
      "displayScore": null,
      "provisionalDisplayScore": 48,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/pokee-isaac-28b",
      "markdownUrl": "https://benchlm.ai/md/models/pokee-isaac-28b.md",
      "id": 405,
      "releaseDate": "2026-08-03",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "pokee-isaac",
        "familyName": "Pokee-Isaac",
        "variantType": "base",
        "snapshotLabel": "v0",
        "baseFamilyModelKey": "pokee-isaac-28b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 48,
        "overallScore": 48,
        "rawOverallScore": 47,
        "verifiedDisplayScore": 48,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": 39.3,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": 39.3,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {
          "agentic": 53
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 5,
        "verifiedBenchmarkCount": 5,
        "rankableBenchmarkCount": 5,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "terminalBench21": 65.1,
          "bfclV4": 70.94,
          "tau3Bench": 66.2,
          "mcpAtlasClaimCoverage": 74.59,
          "pinchBench": 95.67
        },
        "coding": {
          "terminalBench21": 65.1
        },
        "reasoning": {
          "mrcrv2": 60.7
        },
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "gemma-4-26b-a4b",
      "canonicalModelKey": "gemma-4-26b-a4b",
      "model": "Gemma 4 26B A4B",
      "creator": "Google",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": 57.12,
      "provisionalDisplayScore": 46,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 39.05,
        "upper": 75.18
      },
      "rankingEligible": true,
      "overallRank": 93,
      "url": "https://benchlm.ai/models/gemma-4-26b-a4b",
      "markdownUrl": "https://benchlm.ai/md/models/gemma-4-26b-a4b.md",
      "id": 74,
      "releaseDate": "2026-04-02",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gemma-4",
        "familyName": "Gemma 4",
        "variantType": "26b-a4b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gemma-4-31b",
        "relatedModelKeys": [
          "gemma-4-31b",
          "gemma-4-e4b",
          "gemma-4-e2b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 46,
        "overallScore": 46,
        "rawOverallScore": 46,
        "verifiedDisplayScore": 46,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": 38.1,
          "knowledge": 39.1,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": 38.1,
          "knowledge": 39.1,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 93,
        "categoryRanks": {
          "agentic": 67,
          "coding": 108
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 4,
        "verifiedBenchmarkCount": 4,
        "rankableBenchmarkCount": 4,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "aaAgenticIndex": 11.04,
          "tau2Bench": 43.6,
          "gdpvalAaNormalized": 13.5,
          "gdpvalAa": 770
        },
        "coding": {
          "aaCodingIndex": 39.32,
          "aaSciCode": 40
        },
        "reasoning": {
          "lcr": 61.7,
          "critpt": 0
        },
        "multimodalGrounded": {
          "mmmuPro": 73.8,
          "aaMmmuPro": 69.2
        },
        "knowledge": {
          "mmluPro": 82.6,
          "hle": 17.2,
          "hleNoTools": 8.7,
          "artificialAnalysis": 26.07,
          "aaGpqaDiamond": 79.2,
          "aaHle": 19.3,
          "aaOmniscienceIndex": -50.8,
          "omniscienceAccuracy": 19.1,
          "omniscienceHallucinationRate": 86.4
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 72.4
        },
        "math": {}
      }
    },
    {
      "slug": "claude-sonnet-4-5",
      "canonicalModelKey": "claude-sonnet-4-5",
      "model": "Claude Sonnet 4.5",
      "creator": "Anthropic",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": 54.48,
      "provisionalDisplayScore": 46,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 42.96,
        "upper": 65.99
      },
      "rankingEligible": true,
      "overallRank": 106,
      "url": "https://benchlm.ai/models/claude-sonnet-4-5",
      "markdownUrl": "https://benchlm.ai/md/models/claude-sonnet-4-5.md",
      "id": 31,
      "releaseDate": "2025-09-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "claude-sonnet-4-5",
        "familyName": "Claude Sonnet 4.5",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "claude-sonnet-4-5",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 46,
        "overallScore": 46,
        "rawOverallScore": 45,
        "verifiedDisplayScore": 46,
        "displayCategoryScores": {
          "agentic": 33.7,
          "coding": 59.8,
          "reasoning": 19.7,
          "multimodalGrounded": null,
          "knowledge": 74.6,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 34.9
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 33.7,
          "coding": 59.8,
          "reasoning": 19.7,
          "multimodalGrounded": null,
          "knowledge": 74.6,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 34.9
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 106,
        "categoryRanks": {
          "agentic": 80,
          "coding": 70
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 9,
        "verifiedBenchmarkCount": 9,
        "rankableBenchmarkCount": 9,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 50,
          "osWorldVerified": 61.4,
          "vitaBench": 17,
          "gertLabs": 48.51,
          "jobBench": 27.7
        },
        "coding": {
          "sweVerified": 77.2
        },
        "reasoning": {
          "arcAgi2": 13.6
        },
        "multimodalGrounded": {
          "designArenaWebsite": 1208
        },
        "knowledge": {
          "gpqa": 83.4
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {
          "aime2025": 87,
          "frontierMathV2Tiers13": 13.495,
          "frontierMathV2Tier4": 4.167
        }
      }
    },
    {
      "slug": "gemini-3-1-flash-lite",
      "canonicalModelKey": "gemini-3-1-flash-lite",
      "model": "Gemini 3.1 Flash-Lite",
      "creator": "Google",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 50.82,
      "provisionalDisplayScore": 46,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 24.61,
        "upper": 77.03
      },
      "rankingEligible": true,
      "overallRank": 129,
      "url": "https://benchlm.ai/models/gemini-3-1-flash-lite",
      "markdownUrl": "https://benchlm.ai/md/models/gemini-3-1-flash-lite.md",
      "id": 113,
      "releaseDate": "2026-03-03",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gemini-3-1-flash-lite",
        "familyName": "Gemini 3.1 Flash-Lite",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gemini-3-1-flash-lite",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 46,
        "overallScore": 46,
        "rawOverallScore": 45,
        "verifiedDisplayScore": 46,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": 35.2,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": 35.2,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 129,
        "categoryRanks": {
          "coding": 120
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 2,
        "verifiedBenchmarkCount": 2,
        "rankableBenchmarkCount": 2,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "gertLabs": 38.46
        },
        "coding": {
          "vibeCodeBench": 0
        },
        "reasoning": {},
        "multimodalGrounded": {
          "charxiv": 73.2
        },
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "hy3-preview",
      "canonicalModelKey": "hy3-preview",
      "model": "Hy3 Preview",
      "creator": "Tencent",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": 43.71,
      "provisionalDisplayScore": 46,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 33.84,
        "upper": 53.58
      },
      "rankingEligible": true,
      "overallRank": 176,
      "url": "https://benchlm.ai/models/hy3-preview",
      "markdownUrl": "https://benchlm.ai/md/models/hy3-preview.md",
      "id": 111,
      "releaseDate": "2026-04-23",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "hy3",
        "familyName": "Hy3",
        "variantType": "preview",
        "snapshotLabel": "preview",
        "baseFamilyModelKey": "hy3",
        "relatedModelKeys": [
          "hy3"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 46,
        "overallScore": 46,
        "rawOverallScore": 45,
        "verifiedDisplayScore": 46,
        "displayCategoryScores": {
          "agentic": 41.2,
          "coding": 46.2,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 34,
          "multilingual": null,
          "instructionFollowing": 59,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 41.2,
          "coding": 46.2,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 34,
          "multilingual": null,
          "instructionFollowing": 59,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 176,
        "categoryRanks": {
          "agentic": 79,
          "coding": 99,
          "instructionFollowing": 32
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 8,
        "verifiedBenchmarkCount": 8,
        "rankableBenchmarkCount": 8,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 54.4,
          "gertLabs": 36.91,
          "aaAgenticIndex": 31.42,
          "gdpvalAaNormalized": 35.8,
          "gdpvalAa": 1212
        },
        "coding": {
          "sweVerified": 74.4,
          "terminalBench2": 54.4,
          "sciCode": 41.2,
          "aaSciCode": 47.6,
          "aaCodingIndex": 58.8
        },
        "reasoning": {
          "lcr": 66.7,
          "critpt": 4.9
        },
        "multimodalGrounded": {},
        "knowledge": {
          "artificialAnalysis": 41.23,
          "gpqa": 87.2,
          "gpqaDiamond": 87.2,
          "hle": 25.5,
          "omniscienceAccuracy": 31.5,
          "omniscienceHallucinationRate": 73,
          "aaGpqaDiamond": 89.7,
          "aaHle": 33.5,
          "aaOmniscienceIndex": -18.5
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifBench": 63.1
        },
        "math": {}
      }
    },
    {
      "slug": "zaya1-74b-preview",
      "canonicalModelKey": "zaya1-74b-preview",
      "model": "ZAYA1-74B-Preview",
      "creator": "Zyphra",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": null,
      "provisionalDisplayScore": 46,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/zaya1-74b-preview",
      "markdownUrl": "https://benchlm.ai/md/models/zaya1-74b-preview.md",
      "id": 105,
      "releaseDate": "2026-05-07",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "zaya1",
        "familyName": "ZAYA1",
        "variantType": "74b-preview",
        "snapshotLabel": "preview",
        "baseFamilyModelKey": "zaya1-74b-preview",
        "relatedModelKeys": [
          "zaya1-8b"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 46,
        "overallScore": 46,
        "rawOverallScore": 46,
        "verifiedDisplayScore": 46,
        "displayCategoryScores": {
          "agentic": null,
          "coding": 21.3,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 59.9,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 56.5
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": 21.3,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 59.9,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 56.5
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {
          "coding": 84
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 6,
        "verifiedBenchmarkCount": 6,
        "rankableBenchmarkCount": 6,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Airline": 56.1
        },
        "coding": {
          "liveCodeBenchV6": 65.7,
          "sweVerified": 53.2
        },
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {
          "mmluPro": 68.1,
          "gpqa": 57.3,
          "gpqaDiamond": 57.3
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {
          "aime2026": 76.4
        }
      }
    },
    {
      "slug": "qwen3-235b-2507-reasoning",
      "canonicalModelKey": "qwen3-235b-2507-reasoning",
      "model": "Qwen3 235B 2507 (Reasoning)",
      "creator": "Alibaba",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 58.77,
      "provisionalDisplayScore": 45,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 47.26,
        "upper": 70.29
      },
      "rankingEligible": true,
      "overallRank": 83,
      "url": "https://benchlm.ai/models/qwen3-235b-2507-reasoning",
      "markdownUrl": "https://benchlm.ai/md/models/qwen3-235b-2507-reasoning.md",
      "id": 119,
      "releaseDate": "2025-07-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "qwen3-235b-2507",
        "familyName": "Qwen3 235B 2507",
        "variantType": "reasoning",
        "snapshotLabel": "2507",
        "baseFamilyModelKey": "qwen3-235b-2507",
        "relatedModelKeys": [
          "qwen3-235b-2507"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 45,
        "overallScore": 45,
        "rawOverallScore": 44,
        "verifiedDisplayScore": 45,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 30.2
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 30.2
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 83,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 2,
        "verifiedBenchmarkCount": 2,
        "rankableBenchmarkCount": 2,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {
          "frontierMathV2Tiers13": 8.481,
          "frontierMathV2Tier4": 0
        }
      }
    },
    {
      "slug": "gpt-4-1",
      "canonicalModelKey": "gpt-4-1",
      "model": "GPT-4.1",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 50.83,
      "provisionalDisplayScore": 44,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 30.52,
        "upper": 71.13
      },
      "rankingEligible": true,
      "overallRank": 128,
      "url": "https://benchlm.ai/models/gpt-4-1",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-4-1.md",
      "id": 59,
      "releaseDate": "2025-04-14",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-4-1",
        "familyName": "GPT-4.1",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-4-1",
        "relatedModelKeys": [
          "gpt-4-1-mini",
          "gpt-4-1-nano"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "gpt-4o"
      },
      "scores": {
        "displayScore": 44,
        "overallScore": 44,
        "rawOverallScore": 45,
        "verifiedDisplayScore": 44,
        "displayCategoryScores": {
          "agentic": null,
          "coding": 21.3,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 57.5,
          "multilingual": null,
          "instructionFollowing": 72.3,
          "math": 28.1
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": 21.3,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 57.5,
          "multilingual": null,
          "instructionFollowing": 72.3,
          "math": 28.1
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 128,
        "categoryRanks": {
          "coding": 87,
          "instructionFollowing": 30
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 6,
        "verifiedBenchmarkCount": 6,
        "rankableBenchmarkCount": 6,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 47.1,
          "gertLabs": 25.65
        },
        "coding": {
          "sweVerified": 54.6,
          "aaSciCode": 38.1
        },
        "reasoning": {
          "lcr": 64.3,
          "critpt": 0
        },
        "multimodalGrounded": {
          "aaMmmuPro": 61.2,
          "designArenaWebsite": 1057
        },
        "knowledge": {
          "mmlu": 90.2,
          "gpqa": 66.3,
          "artificialAnalysis": 19.61,
          "aaGpqaDiamond": 66.6,
          "aaHle": 4.2,
          "aaOmniscienceIndex": -39.6,
          "omniscienceAccuracy": 27.8,
          "omniscienceHallucinationRate": 93.3
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifeval": 87.4,
          "aaIfBench": 43
        },
        "math": {
          "frontierMathV2Tiers13": 5.517,
          "frontierMathV2Tier4": 0
        },
        "korean": {}
      }
    },
    {
      "slug": "gpt-4-1-mini",
      "canonicalModelKey": "gpt-4-1-mini",
      "model": "GPT-4.1 mini",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 44.4,
      "provisionalDisplayScore": 44,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 32.89,
        "upper": 55.92
      },
      "rankingEligible": true,
      "overallRank": 173,
      "url": "https://benchlm.ai/models/gpt-4-1-mini",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-4-1-mini.md",
      "id": 91,
      "releaseDate": "2025-04-14",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-4-1",
        "familyName": "GPT-4.1",
        "variantType": "mini",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-4-1",
        "relatedModelKeys": [
          "gpt-4-1",
          "gpt-4-1-nano"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": "gpt-4o-mini"
      },
      "scores": {
        "displayScore": 44,
        "overallScore": 44,
        "rawOverallScore": 44,
        "verifiedDisplayScore": 44,
        "displayCategoryScores": {
          "agentic": null,
          "coding": 21.3,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 55.4,
          "multilingual": null,
          "instructionFollowing": 75.5,
          "math": 28.6
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": 21.3,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 55.4,
          "multilingual": null,
          "instructionFollowing": 75.5,
          "math": 28.6
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 173,
        "categoryRanks": {
          "agentic": 104,
          "coding": 112,
          "instructionFollowing": 29
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 5,
        "verifiedBenchmarkCount": 5,
        "rankableBenchmarkCount": 5,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "aaAgenticIndex": 1.79,
          "tau2Bench": 52.9,
          "gdpvalAaNormalized": 0.4,
          "gdpvalAa": 508
        },
        "coding": {
          "sweVerified": 23.6,
          "aaCodingIndex": 20.21,
          "aaSciCode": 40.4
        },
        "reasoning": {
          "lcr": 45.3,
          "critpt": 0
        },
        "multimodalGrounded": {
          "aaMmmuPro": 58.7,
          "designArenaWebsite": 1015
        },
        "knowledge": {
          "mmlu": 87.5,
          "gpqa": 64.2,
          "artificialAnalysis": 14.82,
          "aaGpqaDiamond": 66.4,
          "aaHle": 5,
          "aaOmniscienceIndex": -53.6,
          "omniscienceAccuracy": 20.3,
          "omniscienceHallucinationRate": 92.7
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifeval": 88.5,
          "aaIfBench": 38.3
        },
        "math": {
          "frontierMathV2Tiers13": 4.483
        },
        "korean": {}
      }
    },
    {
      "slug": "grok-4-20-beta",
      "canonicalModelKey": "grok-4-20-beta",
      "model": "Grok 4.20",
      "creator": "xAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "2M",
      "contextWindowTokens": 2000000,
      "displayScore": 55.42,
      "provisionalDisplayScore": 44,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 38.52,
        "upper": 72.32
      },
      "rankingEligible": true,
      "overallRank": 102,
      "url": "https://benchlm.ai/models/grok-4-20-beta",
      "markdownUrl": "https://benchlm.ai/md/models/grok-4-20-beta.md",
      "id": 85,
      "releaseDate": "2026-03-10",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "grok-4-20",
        "familyName": "Grok 4.20",
        "variantType": "reasoning",
        "snapshotLabel": null,
        "baseFamilyModelKey": "grok-4-20-beta",
        "relatedModelKeys": [
          "grok-4-20-multi-agent-beta"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "grok-4-1"
      },
      "scores": {
        "displayScore": 44,
        "overallScore": 44,
        "rawOverallScore": 43,
        "verifiedDisplayScore": 44,
        "displayCategoryScores": {
          "agentic": 30.4,
          "coding": 47.3,
          "reasoning": 52.7,
          "multimodalGrounded": 30.5,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 30.4,
          "coding": 47.3,
          "reasoning": 52.7,
          "multimodalGrounded": 30.5,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 102,
        "categoryRanks": {
          "agentic": 61,
          "coding": 104,
          "multimodalGrounded": 31
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 17,
        "verifiedBenchmarkCount": 17,
        "rankableBenchmarkCount": 17,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 47.1,
          "deepSearchQa": 62.8,
          "gertLabs": 38.36
        },
        "coding": {
          "liveCodeBenchPro": 74.2,
          "sweVerified": 76.7,
          "swePro": 51.8,
          "vibeCodeBench": 4.064
        },
        "reasoning": {
          "arcAgi2": 53.3,
          "arcAgi3": 0.09
        },
        "multimodalGrounded": {
          "mmmuPro": 75.2,
          "charxiv": 60.9,
          "erqa": 54.1,
          "simpleVqa": 57.4,
          "medXpertQaMm": 65.8,
          "designArenaWebsite": 1247
        },
        "knowledge": {
          "gpqaDiamond": 88.5,
          "hleNoTools": 31.6,
          "healthBenchHard": 20.3,
          "medXpertQaText": 50.2
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "gemini-2-5-flash",
      "canonicalModelKey": "gemini-2-5-flash",
      "model": "Gemini 2.5 Flash",
      "creator": "Google",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 47.56,
      "provisionalDisplayScore": 44,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 24.4,
        "upper": 70.73
      },
      "rankingEligible": true,
      "overallRank": 153,
      "url": "https://benchlm.ai/models/gemini-2-5-flash",
      "markdownUrl": "https://benchlm.ai/md/models/gemini-2-5-flash.md",
      "id": 125,
      "releaseDate": "2025-06-17",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gemini-2-5-flash",
        "familyName": "Gemini 2.5 Flash",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gemini-2-5-flash",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 44,
        "overallScore": 44,
        "rawOverallScore": 44,
        "verifiedDisplayScore": 44,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 28.9
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 28.9
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 153,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 2,
        "verifiedBenchmarkCount": 2,
        "rankableBenchmarkCount": 2,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 14.9
        },
        "coding": {
          "aaSciCode": 29.1
        },
        "reasoning": {
          "lcr": 48,
          "critpt": 1.4
        },
        "multimodalGrounded": {
          "aaMmmuPro": 65.5,
          "designArenaWebsite": 1133
        },
        "knowledge": {
          "artificialAnalysis": 14.19,
          "aaGpqaDiamond": 68.3,
          "aaHle": 4.7,
          "aaOmniscienceIndex": -42.6,
          "omniscienceAccuracy": 26.1,
          "omniscienceHallucinationRate": 93
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 39
        },
        "math": {
          "frontierMathV2Tiers13": 4.844,
          "frontierMathV2Tier4": 4.167
        }
      }
    },
    {
      "slug": "qwen3-5-flash",
      "canonicalModelKey": "qwen3-5-flash",
      "model": "Qwen3.5 Flash",
      "creator": "Alibaba",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 47.56,
      "provisionalDisplayScore": 44,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 24.54,
        "upper": 70.59
      },
      "rankingEligible": true,
      "overallRank": 152,
      "url": "https://benchlm.ai/models/qwen3-5-flash",
      "markdownUrl": "https://benchlm.ai/md/models/qwen3-5-flash.md",
      "id": 94,
      "releaseDate": "2026-03-04",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "qwen3-5-flash",
        "familyName": "Qwen3.5 Flash",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "qwen3-5-flash",
        "relatedModelKeys": [
          "qwen3-5-35b-a3b"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 44,
        "overallScore": 44,
        "rawOverallScore": 43,
        "verifiedDisplayScore": 44,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 28.6
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 28.6
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 152,
        "categoryRanks": {
          "coding": 124
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 2,
        "verifiedBenchmarkCount": 2,
        "rankableBenchmarkCount": 2,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {
          "frontierMathV2Tiers13": 6.207,
          "frontierMathV2Tier4": 0
        }
      }
    },
    {
      "slug": "mellum2-12b-a2-5b-instruct",
      "canonicalModelKey": "mellum2-12b-a2-5b-instruct",
      "model": "Mellum2-12B-A2.5B-Instruct",
      "creator": "JetBrains",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": null,
      "provisionalDisplayScore": 44,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/mellum2-12b-a2-5b-instruct",
      "markdownUrl": "https://benchlm.ai/md/models/mellum2-12b-a2-5b-instruct.md",
      "id": 253,
      "releaseDate": "2026-05-28",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "mellum2-12b-a2-5b",
        "familyName": "Mellum2 12B-A2.5B",
        "variantType": "instruct",
        "snapshotLabel": null,
        "baseFamilyModelKey": "mellum2-12b-a2-5b-instruct",
        "relatedModelKeys": [
          "mellum2-12b-a2-5b-thinking"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 44,
        "overallScore": 44,
        "rawOverallScore": 44,
        "verifiedDisplayScore": 44,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 32,
          "multilingual": null,
          "instructionFollowing": 38.4,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 32,
          "multilingual": null,
          "instructionFollowing": 38.4,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {
          "agentic": 78,
          "instructionFollowing": 40
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 5,
        "verifiedBenchmarkCount": 5,
        "rankableBenchmarkCount": 5,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "bfclV4": 44.2
        },
        "coding": {
          "liveCodeBenchV6": 37.2
        },
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {
          "mmluRedux": 78.1,
          "gpqa": 40.9,
          "gpqaDiamond": 40.9
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifeval": 75.8
        },
        "math": {}
      }
    },
    {
      "slug": "laguna-m-1",
      "canonicalModelKey": "laguna-m-1",
      "model": "Laguna M.1",
      "creator": "Poolside",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": null,
      "provisionalDisplayScore": 43,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/laguna-m-1",
      "markdownUrl": "https://benchlm.ai/md/models/laguna-m-1.md",
      "id": 118,
      "releaseDate": "2026-04-28",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "laguna",
        "familyName": "Laguna",
        "variantType": "m-1",
        "snapshotLabel": null,
        "baseFamilyModelKey": "laguna-m-1",
        "relatedModelKeys": [
          "laguna-xs-2"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 43,
        "overallScore": 43,
        "rawOverallScore": 42,
        "verifiedDisplayScore": 43,
        "displayCategoryScores": {
          "agentic": 28.4,
          "coding": 42.8,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 28.4,
          "coding": 42.8,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {
          "agentic": 64,
          "coding": 86
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 5,
        "verifiedBenchmarkCount": 5,
        "rankableBenchmarkCount": 5,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 45.8
        },
        "coding": {
          "sweVerified": 74.6,
          "sweMultilingual": 63.1,
          "swePro": 49.2,
          "terminalBench2": 45.8
        },
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "glm-4-6",
      "canonicalModelKey": "glm-4-6",
      "model": "GLM-4.6",
      "creator": "Z.AI",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": 54.36,
      "provisionalDisplayScore": 43,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 35.64,
        "upper": 73.07
      },
      "rankingEligible": true,
      "overallRank": 107,
      "url": "https://benchlm.ai/models/glm-4-6",
      "markdownUrl": "https://benchlm.ai/md/models/glm-4-6.md",
      "id": 234,
      "releaseDate": "2025-09-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "glm-4-6",
        "familyName": "GLM-4.6",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "glm-4-6",
        "relatedModelKeys": [
          "glm-4-7"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 43,
        "overallScore": 43,
        "rawOverallScore": 43,
        "verifiedDisplayScore": 43,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 27.6
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 27.6
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 107,
        "categoryRanks": {
          "coding": 80
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 3,
        "verifiedBenchmarkCount": 3,
        "rankableBenchmarkCount": 3,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 76.9
        },
        "coding": {
          "vibeCodeBench": 3.09,
          "aaSciCode": 33.1
        },
        "reasoning": {
          "lcr": 28.3,
          "critpt": 0
        },
        "multimodalGrounded": {},
        "knowledge": {
          "artificialAnalysis": 23.38,
          "aaGpqaDiamond": 63.2,
          "aaHle": 5.5,
          "aaOmniscienceIndex": -31.7,
          "omniscienceAccuracy": 21.4,
          "omniscienceHallucinationRate": 67.6
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 36.7
        },
        "math": {
          "frontierMathV2Tiers13": 3.819,
          "frontierMathV2Tier4": 2.128
        }
      }
    },
    {
      "slug": "gemma-4-31b",
      "canonicalModelKey": "gemma-4-31b",
      "model": "Gemma 4 31B",
      "creator": "Google",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": 60.08,
      "provisionalDisplayScore": 43,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 43.34,
        "upper": 76.82
      },
      "rankingEligible": true,
      "overallRank": 71,
      "url": "https://benchlm.ai/models/gemma-4-31b",
      "markdownUrl": "https://benchlm.ai/md/models/gemma-4-31b.md",
      "id": 65,
      "releaseDate": "2026-04-02",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gemma-4",
        "familyName": "Gemma 4",
        "variantType": "31b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gemma-4-31b",
        "relatedModelKeys": [
          "gemma-4-26b-a4b",
          "gemma-4-e4b",
          "gemma-4-e2b"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 43,
        "overallScore": 43,
        "rawOverallScore": 43,
        "verifiedDisplayScore": 43,
        "displayCategoryScores": {
          "agentic": null,
          "coding": 19.3,
          "reasoning": null,
          "multimodalGrounded": 53.3,
          "knowledge": 47,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": 19.3,
          "reasoning": null,
          "multimodalGrounded": 53.3,
          "knowledge": 47,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 71,
        "categoryRanks": {
          "agentic": 136,
          "coding": 93
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 6,
        "verifiedBenchmarkCount": 6,
        "rankableBenchmarkCount": 6,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "aaAgenticIndex": 14.38,
          "tau2Bench": 59.9,
          "gdpvalAaNormalized": 15.7,
          "gdpvalAa": 814,
          "gertLabs": 35.26
        },
        "coding": {
          "sweRebench": 41.6,
          "reactNativeEvals": 75.2,
          "aaCodingIndex": 43.43,
          "aaSciCode": 43.4
        },
        "reasoning": {
          "lcr": 68.3,
          "critpt": 1.4
        },
        "multimodalGrounded": {
          "mmmuPro": 76.9,
          "aaMmmuPro": 73.4
        },
        "knowledge": {
          "gpqa": 84.3,
          "mmluPro": 85.2,
          "hle": 26.5,
          "hleNoTools": 19.5,
          "artificialAnalysis": 29.69,
          "aaGpqaDiamond": 85.7,
          "aaHle": 23.6,
          "aaOmniscienceIndex": -47.9,
          "omniscienceAccuracy": 20,
          "omniscienceHallucinationRate": 85
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 75.6
        },
        "math": {}
      }
    },
    {
      "slug": "grok-3-beta",
      "canonicalModelKey": "grok-3-beta",
      "model": "Grok 3 [Beta]",
      "creator": "xAI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 40.33,
      "provisionalDisplayScore": 43,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 8.65,
        "upper": 72
      },
      "rankingEligible": true,
      "overallRank": 197,
      "url": "https://benchlm.ai/models/grok-3-beta",
      "markdownUrl": "https://benchlm.ai/md/models/grok-3-beta.md",
      "id": 149,
      "releaseDate": "2025-02-19",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "grok-3",
        "familyName": "Grok 3",
        "variantType": "snapshot",
        "snapshotLabel": "beta",
        "baseFamilyModelKey": "grok-3-beta",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 43,
        "overallScore": 43,
        "rawOverallScore": 43,
        "verifiedDisplayScore": 43,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 26.9
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 26.9
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 197,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 2,
        "verifiedBenchmarkCount": 2,
        "rankableBenchmarkCount": 2,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {
          "frontierMathV2Tiers13": 3.793,
          "frontierMathV2Tier4": 0
        }
      }
    },
    {
      "slug": "deepseek-v3",
      "canonicalModelKey": "deepseek-v3",
      "model": "DeepSeek V3",
      "creator": "DeepSeek",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 44.38,
      "provisionalDisplayScore": 43,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 25.19,
        "upper": 63.57
      },
      "rankingEligible": true,
      "overallRank": 174,
      "url": "https://benchlm.ai/models/deepseek-v3",
      "markdownUrl": "https://benchlm.ai/md/models/deepseek-v3.md",
      "id": 112,
      "releaseDate": "2024-12-26",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "deepseek",
        "familyName": "DeepSeek",
        "variantType": "snapshot",
        "snapshotLabel": "V3",
        "baseFamilyModelKey": "deepseek-v3",
        "relatedModelKeys": [
          "deepseek-v3-1",
          "deepseek-v3-2"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 43,
        "overallScore": 43,
        "rawOverallScore": 42,
        "verifiedDisplayScore": 43,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 26.2
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 26.2
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 174,
        "categoryRanks": {
          "agentic": 120,
          "coding": 122
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "aaAgenticIndex": 1.58,
          "tau2Bench": 22.8,
          "gdpvalAaNormalized": 0,
          "gdpvalAa": 235
        },
        "coding": {
          "liveCodeBench": 37.6,
          "sweVerified": 42,
          "aaCodingIndex": 23.04,
          "aaSciCode": 35.4
        },
        "reasoning": {
          "lcr": 31.7,
          "critpt": 0
        },
        "multimodalGrounded": {
          "designArenaWebsite": 1138
        },
        "knowledge": {
          "gpqa": 59.1,
          "mmluPro": 75.9,
          "artificialAnalysis": 14.22,
          "aaGpqaDiamond": 55.7,
          "aaHle": 2.9,
          "aaOmniscienceIndex": -41.6,
          "omniscienceAccuracy": 25.5,
          "omniscienceHallucinationRate": 90
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifeval": 86.1,
          "aaIfBench": 34.8
        },
        "math": {
          "frontierMathV2Tiers13": 1.724
        }
      }
    },
    {
      "slug": "claude-3-5-sonnet",
      "canonicalModelKey": "claude-3-5-sonnet",
      "model": "Claude 3.5 Sonnet",
      "creator": "Anthropic",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": 48.36,
      "provisionalDisplayScore": 42,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 36.85,
        "upper": 59.87
      },
      "rankingEligible": true,
      "overallRank": 144,
      "url": "https://benchlm.ai/models/claude-3-5-sonnet",
      "markdownUrl": "https://benchlm.ai/md/models/claude-3-5-sonnet.md",
      "id": 95,
      "releaseDate": "2024-06-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "claude-3-5-sonnet",
        "familyName": "Claude 3.5 Sonnet",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "claude-3-5-sonnet",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 42,
        "overallScore": 42,
        "rawOverallScore": 42,
        "verifiedDisplayScore": 42,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 25.7
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 25.7
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 144,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 2,
        "verifiedBenchmarkCount": 2,
        "rankableBenchmarkCount": 2,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {
          "sweVerified": 49
        },
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {
          "gpqa": 59.4
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {
          "frontierMathV2Tiers13": 2.069,
          "frontierMathV2Tier4": 0
        }
      }
    },
    {
      "slug": "gpt-4o",
      "canonicalModelKey": "gpt-4o",
      "model": "GPT-4o",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 40.96,
      "provisionalDisplayScore": 42,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 21.58,
        "upper": 60.34
      },
      "rankingEligible": true,
      "overallRank": 191,
      "url": "https://benchlm.ai/models/gpt-4o",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-4o.md",
      "id": 108,
      "releaseDate": "2024-05-13",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-4o",
        "familyName": "GPT-4o",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-4o",
        "relatedModelKeys": [
          "gpt-4o-mini"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 42,
        "overallScore": 42,
        "rawOverallScore": 42,
        "verifiedDisplayScore": 42,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 25.4
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 25.4
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 191,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 25.1
        },
        "coding": {
          "aaSciCode": 33.3
        },
        "reasoning": {
          "lcr": 0,
          "critpt": 0
        },
        "multimodalGrounded": {
          "designArenaWebsite": 849
        },
        "knowledge": {
          "artificialAnalysis": 11.11,
          "aaGpqaDiamond": 54.3,
          "aaHle": 2.4,
          "aaOmniscienceIndex": -10.5,
          "omniscienceAccuracy": 19.9,
          "omniscienceHallucinationRate": 37.9
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 34.3
        },
        "math": {
          "frontierMathV2Tiers13": 0.345
        },
        "korean": {}
      }
    },
    {
      "slug": "llama-4-scout",
      "canonicalModelKey": "llama-4-scout",
      "model": "Llama 4 Scout",
      "creator": "Meta",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "10M",
      "contextWindowTokens": 10000000,
      "displayScore": 39.55,
      "provisionalDisplayScore": 42,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 20.68,
        "upper": 58.41
      },
      "rankingEligible": true,
      "overallRank": 200,
      "url": "https://benchlm.ai/models/llama-4-scout",
      "markdownUrl": "https://benchlm.ai/md/models/llama-4-scout.md",
      "id": 151,
      "releaseDate": "2026-02-28",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "llama-4-scout",
        "familyName": "Llama 4 Scout",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "llama-4-scout",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 42,
        "overallScore": 42,
        "rawOverallScore": 42,
        "verifiedDisplayScore": 42,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 25.4
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 25.4
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 200,
        "categoryRanks": {
          "agentic": 127,
          "coding": 133
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "aaAgenticIndex": 1.1,
          "tau2Bench": 15.5,
          "gdpvalAaNormalized": 0,
          "gdpvalAa": 111
        },
        "coding": {
          "aaCodingIndex": 8.17,
          "aaSciCode": 17
        },
        "reasoning": {
          "lcr": 30.3,
          "critpt": 0
        },
        "multimodalGrounded": {
          "aaMmmuPro": 52.9,
          "designArenaWebsite": 768
        },
        "knowledge": {
          "artificialAnalysis": 10.26,
          "aaGpqaDiamond": 58.7,
          "aaHle": 3.8,
          "aaOmniscienceIndex": -52.1,
          "omniscienceAccuracy": 15.2,
          "omniscienceHallucinationRate": 79.4
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 39.5
        },
        "math": {
          "frontierMathV2Tiers13": 0
        }
      }
    },
    {
      "slug": "llama-4-maverick",
      "canonicalModelKey": "llama-4-maverick",
      "model": "Llama 4 Maverick",
      "creator": "Meta",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 22.76,
      "provisionalDisplayScore": 42,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 15.17,
        "upper": 30.34
      },
      "rankingEligible": true,
      "overallRank": 215,
      "url": "https://benchlm.ai/models/llama-4-maverick",
      "markdownUrl": "https://benchlm.ai/md/models/llama-4-maverick.md",
      "id": 150,
      "releaseDate": "2026-02-28",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "llama-4-maverick",
        "familyName": "Llama 4 Maverick",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "llama-4-maverick",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 42,
        "overallScore": 42,
        "rawOverallScore": 42,
        "verifiedDisplayScore": 42,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 25.4
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 25.4
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 215,
        "categoryRanks": {
          "agentic": 137,
          "coding": 144
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "aaAgenticIndex": 1.24,
          "tau2Bench": 17.8,
          "gdpvalAaNormalized": 0,
          "gdpvalAa": 7
        },
        "coding": {
          "aaCodingIndex": 16.28,
          "aaSciCode": 33.1
        },
        "reasoning": {
          "lcr": 50,
          "critpt": 0
        },
        "multimodalGrounded": {
          "aaMmmuPro": 62.1,
          "designArenaWebsite": 889
        },
        "knowledge": {
          "artificialAnalysis": 14.48,
          "aaGpqaDiamond": 67.1,
          "aaHle": 4.9,
          "aaOmniscienceIndex": -41.8,
          "omniscienceAccuracy": 24.9,
          "omniscienceHallucinationRate": 88.9
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 43
        },
        "math": {
          "frontierMathV2Tiers13": 0.69
        }
      }
    },
    {
      "slug": "ling-2-6-flash",
      "canonicalModelKey": "ling-2-6-flash",
      "model": "Ling 2.6 Flash",
      "creator": "InclusionAI",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": 44.18,
      "provisionalDisplayScore": 42,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 32.66,
        "upper": 55.69
      },
      "rankingEligible": true,
      "overallRank": 175,
      "url": "https://benchlm.ai/models/ling-2-6-flash",
      "markdownUrl": "https://benchlm.ai/md/models/ling-2-6-flash.md",
      "id": 153,
      "releaseDate": "2026-04-21",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "ling-2-6",
        "familyName": "Ling 2.6",
        "variantType": "flash",
        "snapshotLabel": null,
        "baseFamilyModelKey": "ling-2-6-flash",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 42,
        "overallScore": 42,
        "rawOverallScore": 42,
        "verifiedDisplayScore": 42,
        "displayCategoryScores": {
          "agentic": null,
          "coding": 21.4,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 50.1,
          "multilingual": null,
          "instructionFollowing": 48.2,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": 21.4,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 50.1,
          "multilingual": null,
          "instructionFollowing": 48.2,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 175,
        "categoryRanks": {
          "agentic": 103,
          "coding": 106,
          "instructionFollowing": 37
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 3,
        "verifiedBenchmarkCount": 3,
        "rankableBenchmarkCount": 3,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 86,
          "gdpvalAaNormalized": 2.3,
          "gdpvalAa": 550,
          "aaAgenticIndex": 2.25
        },
        "coding": {
          "sciCode": 27,
          "aaCodingIndex": 25.26,
          "aaSciCode": 27.1
        },
        "reasoning": {
          "lcr": 28,
          "critpt": 0
        },
        "multimodalGrounded": {},
        "knowledge": {
          "artificialAnalysis": 14.05,
          "gpqa": 59,
          "aaGpqaDiamond": 59.3,
          "aaHle": 6.3,
          "aaOmniscienceIndex": -66.1,
          "omniscienceAccuracy": 15.6,
          "omniscienceHallucinationRate": 96.7
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifBench": 57,
          "aaIfBench": 57.4
        },
        "math": {}
      }
    },
    {
      "slug": "gemma-4-12b",
      "canonicalModelKey": "gemma-4-12b",
      "model": "Gemma 4 12B",
      "creator": "Google",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": 47.33,
      "provisionalDisplayScore": 41,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 35.82,
        "upper": 58.85
      },
      "rankingEligible": true,
      "overallRank": 154,
      "url": "https://benchlm.ai/models/gemma-4-12b",
      "markdownUrl": "https://benchlm.ai/md/models/gemma-4-12b.md",
      "id": 251,
      "releaseDate": "2026-06-03",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gemma-4",
        "familyName": "Gemma 4",
        "variantType": "12b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gemma-4-31b",
        "relatedModelKeys": [
          "gemma-4-31b",
          "gemma-4-26b-a4b",
          "gemma-4-e4b",
          "gemma-4-e2b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 41,
        "overallScore": 41,
        "rawOverallScore": 41,
        "verifiedDisplayScore": 41,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": 16.7,
          "multimodalGrounded": 15.1,
          "knowledge": 72.8,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 57.7
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": 16.7,
          "multimodalGrounded": 15.1,
          "knowledge": 72.8,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 57.7
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 154,
        "categoryRanks": {
          "coding": 92
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 11,
        "verifiedBenchmarkCount": 11,
        "rankableBenchmarkCount": 11,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 36.3,
          "gdpvalAaNormalized": 7.3,
          "gdpvalAa": 647
        },
        "coding": {
          "liveCodeBenchV6": 72,
          "aaSciCode": 38.2,
          "aaCodingIndex": 30.96
        },
        "reasoning": {
          "bbh": 53,
          "mrcrv2": 43.4,
          "lcr": 61.7,
          "critpt": 0
        },
        "multimodalGrounded": {
          "mmmuPro": 69.1,
          "mathVision": 79.7,
          "medXpertQaMm": 48.7,
          "aaMmmuPro": 69.7
        },
        "knowledge": {
          "gpqa": 78.8,
          "gpqaDiamond": 78.8,
          "mmluPro": 77.2,
          "hleNoTools": 5.2,
          "mmmlu": 83.4,
          "artificialAnalysis": 22.16,
          "aaGpqaDiamond": 75.3,
          "aaHle": 15.7,
          "aaOmniscienceIndex": -52.7,
          "omniscienceAccuracy": 15.6,
          "omniscienceHallucinationRate": 81
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 73.5
        },
        "math": {
          "aime2026": 77.5
        }
      }
    },
    {
      "slug": "gemini-2-5-pro",
      "canonicalModelKey": "gemini-2-5-pro",
      "model": "Gemini 2.5 Pro",
      "creator": "Google",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 56.77,
      "provisionalDisplayScore": 41,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 38.38,
        "upper": 75.17
      },
      "rankingEligible": true,
      "overallRank": 94,
      "url": "https://benchlm.ai/models/gemini-2-5-pro",
      "markdownUrl": "https://benchlm.ai/md/models/gemini-2-5-pro.md",
      "id": 68,
      "releaseDate": "2025-03-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gemini-2-5-pro",
        "familyName": "Gemini 2.5 Pro",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gemini-2-5-pro",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 41,
        "overallScore": 41,
        "rawOverallScore": 40,
        "verifiedDisplayScore": 41,
        "displayCategoryScores": {
          "agentic": null,
          "coding": 37,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 21.7,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 35.4
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": 37,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 21.7,
          "multilingual": null,
          "instructionFollowing": null,
          "math": 35.4
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 94,
        "categoryRanks": {
          "agentic": 76,
          "coding": 136
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 5,
        "verifiedBenchmarkCount": 5,
        "rankableBenchmarkCount": 5,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "aaAgenticIndex": 7.24,
          "tau2Bench": 54.1,
          "gertLabs": 42.01,
          "gdpvalAaNormalized": 8.7,
          "gdpvalAa": 673
        },
        "coding": {
          "sweVerified": 63.8,
          "vibeCodeBench": 0.4,
          "aaCodingIndex": 33.25,
          "aaSciCode": 42.8
        },
        "reasoning": {
          "lcr": 66,
          "critpt": 2.6
        },
        "multimodalGrounded": {
          "aaMmmuPro": 74.9,
          "designArenaWebsite": 1185
        },
        "knowledge": {
          "gpqa": 83,
          "hle": 18.8,
          "artificialAnalysis": 25.91,
          "aaGpqaDiamond": 84.4,
          "aaHle": 22.5,
          "aaOmniscienceIndex": -16.3,
          "omniscienceAccuracy": 39.1,
          "omniscienceHallucinationRate": 90.9
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 48.7
        },
        "math": {
          "frontierMathV2Tiers13": 14.138,
          "frontierMathV2Tier4": 4.167
        }
      }
    },
    {
      "slug": "minicpm5-1b",
      "canonicalModelKey": "minicpm5-1b",
      "model": "MiniCPM5-1B",
      "creator": "OpenBMB",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "131K",
      "contextWindowTokens": 131000,
      "displayScore": 11.08,
      "provisionalDisplayScore": 40,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 1.21,
        "upper": 20.95
      },
      "rankingEligible": true,
      "overallRank": 225,
      "url": "https://benchlm.ai/models/minicpm5-1b",
      "markdownUrl": "https://benchlm.ai/md/models/minicpm5-1b.md",
      "id": 123,
      "releaseDate": "2026-05-25",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "minicpm5",
        "familyName": "MiniCPM5",
        "variantType": "1b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "minicpm5-1b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 40,
        "overallScore": 40,
        "rawOverallScore": 40,
        "verifiedDisplayScore": 40,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 40.8,
          "multilingual": null,
          "instructionFollowing": 24.7,
          "math": 5.5
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 40.8,
          "multilingual": null,
          "instructionFollowing": 24.7,
          "math": 5.5
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 225,
        "categoryRanks": {
          "agentic": 134,
          "instructionFollowing": 42
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 13,
        "verifiedBenchmarkCount": 13,
        "rankableBenchmarkCount": 13,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "bfclV4": 25.15
        },
        "coding": {
          "liveCodeBenchPro": 22.68,
          "liveCodeBenchV6": 33.52
        },
        "reasoning": {
          "bbh": 71.89
        },
        "multimodalGrounded": {},
        "knowledge": {
          "mmluPro": 48.85,
          "mmluRedux": 70.06,
          "gpqaDiamond": 26.26,
          "superGpqa": 23.14
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifBench": 46.67,
          "ifeval": 80.41
        },
        "math": {
          "aime2025": 40.42,
          "aime2026": 40.42,
          "hmmtFeb2026": 25.76,
          "math500": 91.6
        }
      }
    },
    {
      "slug": "ornith-1-0-9b",
      "canonicalModelKey": "ornith-1-0-9b",
      "model": "Ornith-1.0-9B",
      "creator": "DeepReinforce AI",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": null,
      "provisionalDisplayScore": 40,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/ornith-1-0-9b",
      "markdownUrl": "https://benchlm.ai/md/models/ornith-1-0-9b.md",
      "id": 270,
      "releaseDate": "2026-06-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "ornith-1-0",
        "familyName": "Ornith 1.0",
        "variantType": "9b",
        "snapshotLabel": "9B",
        "baseFamilyModelKey": "ornith-1-0-397b",
        "relatedModelKeys": [
          "ornith-1-0-397b",
          "ornith-1-0-35b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 40,
        "overallScore": 40,
        "rawOverallScore": 39,
        "verifiedDisplayScore": 40,
        "displayCategoryScores": {
          "agentic": 24.4,
          "coding": 36.9,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 24.4,
          "coding": 36.9,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {
          "agentic": 65,
          "coding": 79
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 6,
        "verifiedBenchmarkCount": 6,
        "rankableBenchmarkCount": 6,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 43.1,
          "clawEval": 63.1
        },
        "coding": {
          "sweVerified": 69.4,
          "swePro": 42.9,
          "sweMultilingual": 52,
          "nl2Repo": 27.2,
          "terminalBench2": 43.1
        },
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "laguna-xs-2",
      "canonicalModelKey": "laguna-xs-2",
      "model": "Laguna XS.2",
      "creator": "Poolside",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": null,
      "provisionalDisplayScore": 39,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/laguna-xs-2",
      "markdownUrl": "https://benchlm.ai/md/models/laguna-xs-2.md",
      "id": 147,
      "releaseDate": "2026-04-28",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "laguna",
        "familyName": "Laguna",
        "variantType": "xs-2",
        "snapshotLabel": null,
        "baseFamilyModelKey": "laguna-m-1",
        "relatedModelKeys": [
          "laguna-m-1"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 39,
        "overallScore": 39,
        "rawOverallScore": 38,
        "verifiedDisplayScore": 39,
        "displayCategoryScores": {
          "agentic": 21.3,
          "coding": 37.5,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 21.3,
          "coding": 37.5,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {
          "agentic": 74,
          "coding": 97
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 5,
        "verifiedBenchmarkCount": 5,
        "rankableBenchmarkCount": 5,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 35.7
        },
        "coding": {
          "sweVerified": 69.9,
          "sweMultilingual": 57.7,
          "swePro": 46.3,
          "terminalBench2": 35.7
        },
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "ornith-1-5-9b",
      "canonicalModelKey": "ornith-1-5-9b",
      "model": "Ornith-1.5-9B",
      "creator": "Ornith AI",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": 37.33,
      "provisionalDisplayScore": 39,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 27.46,
        "upper": 47.2
      },
      "rankingEligible": true,
      "overallRank": 205,
      "url": "https://benchlm.ai/models/ornith-1-5-9b",
      "markdownUrl": "https://benchlm.ai/md/models/ornith-1-5-9b.md",
      "id": 396,
      "releaseDate": "2026-08-18",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "ornith-1-5",
        "familyName": "Ornith 1.5",
        "variantType": "9b",
        "snapshotLabel": "9B",
        "baseFamilyModelKey": "ornith-1-5-397b",
        "relatedModelKeys": [
          "ornith-1-5-397b",
          "ornith-1-5-35b-a3b",
          "ornith-1-0-9b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": "ornith-1-0-9b"
      },
      "scores": {
        "displayScore": 39,
        "overallScore": 39,
        "rawOverallScore": 38,
        "verifiedDisplayScore": 39,
        "displayCategoryScores": {
          "agentic": 28.1,
          "coding": 38.3,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 27.3,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 28.1,
          "coding": 38.3,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 27.3,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 205,
        "categoryRanks": {
          "agentic": 110,
          "coding": 114
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 12,
        "verifiedBenchmarkCount": 12,
        "rankableBenchmarkCount": 12,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "terminalBench21": 46.2,
          "hleWithTools": 30.5,
          "mcpAtlas": 54.2,
          "toolathlonVerified": 41.2,
          "wideResearch": 59.5,
          "browseComp": 56.4,
          "clawEval": 66.5
        },
        "coding": {
          "terminalBench21": 46.2,
          "sweVerified": 70.6,
          "swePro": 47.5,
          "sweMultilingual": 54.4,
          "nl2Repo": 32.4
        },
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {
          "gpqa": 86.4,
          "gpqaDiamond": 86.4,
          "hle": 20.2,
          "hleNoTools": 20.2
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "nemotron-3-5-lightning-30b-a3b-nvfp4",
      "canonicalModelKey": "nemotron-3-5-lightning-30b-a3b-nvfp4",
      "model": "Nemotron 3.5 Lightning 30B A3B NVFP4",
      "creator": "NVIDIA",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 26.55,
      "provisionalDisplayScore": 37,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 16.68,
        "upper": 36.42
      },
      "rankingEligible": true,
      "overallRank": 212,
      "url": "https://benchlm.ai/models/nemotron-3-5-lightning-30b-a3b-nvfp4",
      "markdownUrl": "https://benchlm.ai/md/models/nemotron-3-5-lightning-30b-a3b-nvfp4.md",
      "id": 386,
      "releaseDate": "2026-08-11",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "nemotron-3-5-lightning",
        "familyName": "Nemotron 3.5 Lightning",
        "variantType": "30b-a3b-nvfp4",
        "snapshotLabel": "NVFP4",
        "baseFamilyModelKey": "nemotron-3-5-lightning-30b-a3b-nvfp4",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 37,
        "overallScore": 37,
        "rawOverallScore": 37,
        "verifiedDisplayScore": 37,
        "displayCategoryScores": {
          "agentic": 11.5,
          "coding": 16.2,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 75.6,
          "multilingual": null,
          "instructionFollowing": 76.3,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": 11.5,
          "coding": 16.2,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 75.6,
          "multilingual": null,
          "instructionFollowing": 76.3,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 212,
        "categoryRanks": {
          "agentic": 131,
          "coding": 134,
          "instructionFollowing": 28
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 13,
        "verifiedBenchmarkCount": 13,
        "rankableBenchmarkCount": 13,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 2
      },
      "benchmarks": {
        "agentic": {
          "terminalBench2": 23.46,
          "pinchBench": 83.43,
          "browseComp": 36.81,
          "tau3Bench": 9.48,
          "gdpvalAa": 865,
          "aaAgenticIndex": 13.76,
          "gdpvalAaNormalized": 16.2
        },
        "coding": {
          "sweVerified": 52.8,
          "sweMultilingual": 36.47,
          "terminalBench2": 23.46,
          "sciCode": 31.38,
          "aaCodingIndex": 26.76,
          "aaSciCode": 31.6
        },
        "reasoning": {
          "lcr": 49.19,
          "critpt": 0
        },
        "multimodalGrounded": {},
        "knowledge": {
          "gpqa": 75.57,
          "gpqaDiamond": 75.57,
          "hleNoTools": 10.47,
          "mmluPro": 81.62,
          "artificialAnalysis": 23.61,
          "aaGpqaDiamond": 74.3,
          "aaHle": 10.6,
          "aaOmniscienceIndex": -17.7,
          "omniscienceAccuracy": 14.4,
          "omniscienceHallucinationRate": 37.6,
          "aaOpennessIndex": 77.8
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifBench": 72.88
        },
        "math": {}
      }
    },
    {
      "slug": "lfm2-5-vl-450m",
      "canonicalModelKey": "lfm2-5-vl-450m",
      "model": "LFM2.5-VL-450M",
      "creator": "LiquidAI",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": null,
      "provisionalDisplayScore": 35,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/lfm2-5-vl-450m",
      "markdownUrl": "https://benchlm.ai/md/models/lfm2-5-vl-450m.md",
      "id": 168,
      "releaseDate": "2026-04-08",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "lfm2-5-vl-450m",
        "familyName": "LFM2.5-VL-450M",
        "variantType": "vl",
        "snapshotLabel": null,
        "baseFamilyModelKey": "lfm2-5-vl-450m",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 35,
        "overallScore": 35,
        "rawOverallScore": 35,
        "verifiedDisplayScore": 35,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 16.1,
          "multilingual": null,
          "instructionFollowing": 26.4,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 16.1,
          "multilingual": null,
          "instructionFollowing": 26.4,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {
          "agentic": 86,
          "instructionFollowing": 41
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 7,
        "verifiedBenchmarkCount": 7,
        "rankableBenchmarkCount": 7,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "bfclV4": 21.08
        },
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {
          "mmmu": 32.67,
          "realWorldQa": 58.43,
          "countBench": 73.31
        },
        "knowledge": {
          "gpqa": 25.66,
          "mmluPro": 19.32
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifeval": 61.16
        },
        "math": {}
      }
    },
    {
      "slug": "command-a-plus",
      "canonicalModelKey": "command-a-plus",
      "model": "Command A+",
      "creator": "Cohere",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 47.57,
      "provisionalDisplayScore": 31,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 36.06,
        "upper": 59.09
      },
      "rankingEligible": true,
      "overallRank": 151,
      "url": "https://benchlm.ai/models/command-a-plus",
      "markdownUrl": "https://benchlm.ai/md/models/command-a-plus.md",
      "id": 86,
      "releaseDate": "2026-05-20",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "command-a",
        "familyName": "Command A",
        "variantType": "plus",
        "snapshotLabel": null,
        "baseFamilyModelKey": "command-a-plus",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 31,
        "overallScore": 31,
        "rawOverallScore": 31,
        "verifiedDisplayScore": 31,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": 8,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": 8,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 151,
        "categoryRanks": {
          "agentic": 92,
          "coding": 94,
          "multimodalGrounded": 33
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": true,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 3,
        "verifiedBenchmarkCount": 3,
        "rankableBenchmarkCount": 3,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 85,
          "aaAgenticIndex": 9.23,
          "gdpvalAaNormalized": 10.7,
          "gdpvalAa": 715,
          "terminalBenchHard": 25
        },
        "coding": {
          "aaCodingIndex": 27.85,
          "aaSciCode": 37.8
        },
        "reasoning": {
          "lcr": 48.7,
          "critpt": 0.3
        },
        "multimodalGrounded": {
          "mmmu": 75.1,
          "mmmuPro": 63,
          "charxiv": 52.7,
          "aaMmmuPro": 63.2
        },
        "knowledge": {
          "artificialAnalysis": 22.51,
          "aaGpqaDiamond": 76.1,
          "aaHle": 12,
          "aaOmniscienceIndex": -4,
          "omniscienceAccuracy": 8.9,
          "omniscienceHallucinationRate": 14.2,
          "aaOpennessIndex": 38.9
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 73.9
        },
        "math": {}
      }
    },
    {
      "slug": "lfm2-5-230m",
      "canonicalModelKey": "lfm2-5-230m",
      "model": "LFM2.5-230M",
      "creator": "LiquidAI",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": null,
      "provisionalDisplayScore": 31,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/lfm2-5-230m",
      "markdownUrl": "https://benchlm.ai/md/models/lfm2-5-230m.md",
      "id": 271,
      "releaseDate": "2026-06-25",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "lfm2-5-230m",
        "familyName": "LFM2.5-230M",
        "variantType": "instruct",
        "snapshotLabel": null,
        "baseFamilyModelKey": "lfm2-5-230m",
        "relatedModelKeys": [
          "lfm2-5-350m",
          "lfm2-5-8b-a1b"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 31,
        "overallScore": 31,
        "rawOverallScore": 31,
        "verifiedDisplayScore": 31,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 16.1,
          "multilingual": null,
          "instructionFollowing": 1,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 16.1,
          "multilingual": null,
          "instructionFollowing": 1,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {
          "agentic": 85,
          "instructionFollowing": 43
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": true,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 6,
        "verifiedBenchmarkCount": 6,
        "rankableBenchmarkCount": 6,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "bfclV4": 21.03
        },
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {
          "gpqa": 25.41,
          "gpqaDiamond": 25.41,
          "mmluPro": 20.25
        },
        "multilingual": {},
        "instructionFollowing": {
          "ifeval": 71.71,
          "ifBench": 38.4
        },
        "math": {}
      }
    },
    {
      "slug": "claude-opus-4-6-thinking",
      "canonicalModelKey": "claude-opus-4-6-thinking",
      "model": "Claude Opus 4.6 (Adaptive)",
      "creator": "Anthropic",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 65.03,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 53.52,
        "upper": 76.55
      },
      "rankingEligible": true,
      "overallRank": 38,
      "url": "https://benchlm.ai/models/claude-opus-4-6-thinking",
      "markdownUrl": "https://benchlm.ai/md/models/claude-opus-4-6-thinking.md",
      "id": 228,
      "releaseDate": "2026-02-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "claude-opus-4-6",
        "familyName": "Claude Opus 4.6",
        "variantType": "reasoning",
        "snapshotLabel": "adaptive",
        "baseFamilyModelKey": "claude-opus-4-6",
        "relatedModelKeys": [
          "claude-opus-4-6"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 38,
        "categoryRanks": {
          "agentic": 29,
          "coding": 35
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "apexAgentsAa": 33,
          "tau2Bench": 92.1
        },
        "coding": {
          "vibeCodeBench": 53.498,
          "aaSciCode": 51.9
        },
        "reasoning": {
          "lcr": 74.3,
          "critpt": 12.6
        },
        "multimodalGrounded": {
          "aaMmmuPro": 75.4,
          "designArenaWebsite": 1309
        },
        "knowledge": {
          "artificialAnalysis": 44.93,
          "aaGpqaDiamond": 89.6,
          "aaHle": 39.9,
          "aaOmniscienceIndex": 13.7,
          "omniscienceAccuracy": 47,
          "omniscienceHallucinationRate": 62.8
        },
        "multilingual": {
          "aaGlobalMmluLite": 92.2
        },
        "instructionFollowing": {
          "aaIfBench": 53.1
        },
        "math": {}
      }
    },
    {
      "slug": "grok-4-1",
      "canonicalModelKey": "grok-4-1",
      "model": "Grok 4.1",
      "creator": "xAI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 60.73,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 49.22,
        "upper": 72.24
      },
      "rankingEligible": true,
      "overallRank": 65,
      "url": "https://benchlm.ai/models/grok-4-1",
      "markdownUrl": "https://benchlm.ai/md/models/grok-4-1.md",
      "id": 7,
      "releaseDate": "2025-11-17",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "grok-4-1",
        "familyName": "Grok 4.1",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "grok-4-1",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 65,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "researchClawBench": 13.5
        },
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "hy3",
      "canonicalModelKey": "hy3",
      "model": "Hy3",
      "creator": "Tencent",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": 67.65,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 58.81,
        "upper": 76.49
      },
      "rankingEligible": true,
      "overallRank": 24,
      "url": "https://benchlm.ai/models/hy3",
      "markdownUrl": "https://benchlm.ai/md/models/hy3.md",
      "id": 273,
      "releaseDate": "2026-07-06",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "hy3",
        "familyName": "Hy3",
        "variantType": "instruct",
        "snapshotLabel": null,
        "baseFamilyModelKey": "hy3",
        "relatedModelKeys": [
          "hy3-preview"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "hy3-preview"
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 24,
        "categoryRanks": {
          "agentic": 94,
          "coding": 23
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "aaAgenticIndex": 31.42,
          "gdpvalAaNormalized": 35.6,
          "gdpvalAa": 1212
        },
        "coding": {
          "aaSciCode": 47.6,
          "aaCodingIndex": 58.8
        },
        "reasoning": {
          "lcr": 74.7,
          "critpt": 4.9
        },
        "multimodalGrounded": {
          "designArenaWebsite": 1200
        },
        "knowledge": {
          "artificialAnalysis": 42.21,
          "aaGpqaDiamond": 89.7,
          "aaHle": 33.5,
          "aaOmniscienceIndex": -18.5,
          "omniscienceAccuracy": 32,
          "omniscienceHallucinationRate": 74.1
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "glm-5-reasoning",
      "canonicalModelKey": "glm-5-reasoning",
      "model": "GLM-5 (Reasoning)",
      "creator": "Z.AI",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": 60.55,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 49.04,
        "upper": 72.07
      },
      "rankingEligible": true,
      "overallRank": 67,
      "url": "https://benchlm.ai/models/glm-5-reasoning",
      "markdownUrl": "https://benchlm.ai/md/models/glm-5-reasoning.md",
      "id": 22,
      "releaseDate": "2026-03-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "glm-5",
        "familyName": "GLM-5",
        "variantType": "reasoning",
        "snapshotLabel": null,
        "baseFamilyModelKey": "glm-5",
        "relatedModelKeys": [
          "glm-5"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 67,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {
          "vibeCodeBench": 23.359
        },
        "reasoning": {},
        "multimodalGrounded": {
          "designArenaWebsite": 1266
        },
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "glm-5-turbo",
      "canonicalModelKey": "glm-5-turbo",
      "model": "GLM-5-Turbo",
      "creator": "Z.AI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": 65.82,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 54.91,
        "upper": 76.74
      },
      "rankingEligible": true,
      "overallRank": 33,
      "url": "https://benchlm.ai/models/glm-5-turbo",
      "markdownUrl": "https://benchlm.ai/md/models/glm-5-turbo.md",
      "id": 204,
      "releaseDate": "2026-03-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "glm-5",
        "familyName": "GLM-5",
        "variantType": "turbo",
        "snapshotLabel": null,
        "baseFamilyModelKey": "glm-5",
        "relatedModelKeys": [
          "glm-5",
          "glm-5-reasoning"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 33,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "clawEval": 55.8,
          "tau2Bench": 98.5
        },
        "coding": {
          "aaSciCode": 43.6
        },
        "reasoning": {
          "lcr": 66.7,
          "critpt": 0.3
        },
        "multimodalGrounded": {
          "designArenaWebsite": 1286
        },
        "knowledge": {
          "artificialAnalysis": 39.05,
          "aaGpqaDiamond": 84.7,
          "aaHle": 27.8,
          "aaOmniscienceIndex": -16.4,
          "omniscienceAccuracy": 28.4,
          "omniscienceHallucinationRate": 62.6
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 73.2
        },
        "math": {}
      }
    },
    {
      "slug": "qwen3-5-397b-reasoning",
      "canonicalModelKey": "qwen3-5-397b-reasoning",
      "model": "Qwen3.5 397B (Reasoning)",
      "creator": "Alibaba",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 60.27,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 48.76,
        "upper": 71.78
      },
      "rankingEligible": true,
      "overallRank": 68,
      "url": "https://benchlm.ai/models/qwen3-5-397b-reasoning",
      "markdownUrl": "https://benchlm.ai/md/models/qwen3-5-397b-reasoning.md",
      "id": 28,
      "releaseDate": "2026-02-16",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "qwen3-5-397b",
        "familyName": "Qwen3.5 397B",
        "variantType": "reasoning",
        "snapshotLabel": null,
        "baseFamilyModelKey": "qwen3-5-397b",
        "relatedModelKeys": [
          "qwen3-5-397b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 68,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "aaAgenticIndex": 19.85,
          "apexAgentsAa": 15.3,
          "tau2Bench": 95.6,
          "gdpvalAaNormalized": 23.3,
          "gdpvalAa": 966
        },
        "coding": {
          "aaCodingIndex": 48.21,
          "aaSciCode": 42
        },
        "reasoning": {
          "lcr": 72.7,
          "critpt": 1.7
        },
        "multimodalGrounded": {
          "aaMmmuPro": 77.3
        },
        "knowledge": {
          "artificialAnalysis": 34.26,
          "aaGpqaDiamond": 89.3,
          "aaHle": 29,
          "aaOmniscienceIndex": -30.7,
          "omniscienceAccuracy": 30.8,
          "omniscienceHallucinationRate": 88.9
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 78.8
        },
        "math": {}
      }
    },
    {
      "slug": "mimo-v2-pro",
      "canonicalModelKey": "mimo-v2-pro",
      "model": "MiMo-V2-Pro",
      "creator": "Xiaomi",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 66.65,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 57.84,
        "upper": 75.46
      },
      "rankingEligible": true,
      "overallRank": 30,
      "url": "https://benchlm.ai/models/mimo-v2-pro",
      "markdownUrl": "https://benchlm.ai/md/models/mimo-v2-pro.md",
      "id": 6,
      "releaseDate": "2026-03-18",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "mimo-v2-pro",
        "familyName": "MiMo-V2-Pro",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "mimo-v2-pro",
        "relatedModelKeys": [
          "mimo-v2-flash",
          "mimo-v2-omni"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 30,
        "categoryRanks": {
          "coding": 33
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "clawEval": 57.8,
          "tau2Bench": 95,
          "gertLabs": 36.68,
          "researchClawBench": 15.3
        },
        "coding": {
          "sweVerified": 78,
          "aaSciCode": 42.5
        },
        "reasoning": {
          "lcr": 65.7,
          "critpt": 0.3
        },
        "multimodalGrounded": {},
        "knowledge": {
          "artificialAnalysis": 41.37,
          "aaGpqaDiamond": 87,
          "aaHle": 30.4,
          "aaOmniscienceIndex": 4.6,
          "omniscienceAccuracy": 26.6,
          "omniscienceHallucinationRate": 30
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 68.8
        },
        "math": {}
      }
    },
    {
      "slug": "gpt-5-2-instant",
      "canonicalModelKey": "gpt-5-2-instant",
      "model": "GPT-5.2 Instant",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 59.77,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 48.26,
        "upper": 71.29
      },
      "rankingEligible": true,
      "overallRank": 72,
      "url": "https://benchlm.ai/models/gpt-5-2-instant",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-5-2-instant.md",
      "id": 179,
      "releaseDate": "2025-12-11",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-5-2",
        "familyName": "GPT-5.2",
        "variantType": "instant",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-5-2",
        "relatedModelKeys": [
          "gpt-5-2",
          "gpt-5-2-pro"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 72,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 2,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "gpt-5-2-pro",
      "canonicalModelKey": "gpt-5-2-pro",
      "model": "GPT-5.2 Pro",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "400K",
      "contextWindowTokens": 400000,
      "displayScore": 67.19,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 56.87,
        "upper": 77.52
      },
      "rankingEligible": true,
      "overallRank": 27,
      "url": "https://benchlm.ai/models/gpt-5-2-pro",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-5-2-pro.md",
      "id": 177,
      "releaseDate": "2025-12-11",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-5-2",
        "familyName": "GPT-5.2",
        "variantType": "pro",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-5-2",
        "relatedModelKeys": [
          "gpt-5-2",
          "gpt-5-2-instant"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 27,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 4,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "gpt-5-3-instant",
      "canonicalModelKey": "gpt-5-3-instant",
      "model": "GPT-5.3 Instant",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "400K",
      "contextWindowTokens": 400000,
      "displayScore": 59.67,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 48.15,
        "upper": 71.18
      },
      "rankingEligible": true,
      "overallRank": 75,
      "url": "https://benchlm.ai/models/gpt-5-3-instant",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-5-3-instant.md",
      "id": 178,
      "releaseDate": "2026-03-03",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-5-3",
        "familyName": "GPT-5.3",
        "variantType": "instant",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-5-3-instant",
        "relatedModelKeys": [
          "gpt-5-3-codex"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 75,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 5,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "gpt-5-high",
      "canonicalModelKey": "gpt-5-high",
      "model": "GPT-5 (high)",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 59.4,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 47.88,
        "upper": 70.91
      },
      "rankingEligible": true,
      "overallRank": 80,
      "url": "https://benchlm.ai/models/gpt-5-high",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-5-high.md",
      "id": 19,
      "releaseDate": "2025-08-07",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-5",
        "familyName": "GPT-5",
        "variantType": "reasoning",
        "snapshotLabel": "high",
        "baseFamilyModelKey": "gpt-5-high",
        "relatedModelKeys": [
          "gpt-5-medium",
          "gpt-5-mini",
          "gpt-5-nano"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 80,
        "categoryRanks": {
          "agentic": 59
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 2,
        "verifiedBenchmarkCount": 2,
        "rankableBenchmarkCount": 2,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "aaAgenticIndex": 26.53,
          "tau2Bench": 84.8,
          "gdpvalAaNormalized": 29.1,
          "gdpvalAa": 1082,
          "jobBench": 8.53
        },
        "coding": {
          "vibeCodeBench": 20.088,
          "aaCodingIndex": 37.78,
          "aaSciCode": 42.9
        },
        "reasoning": {
          "lcr": 76.3,
          "critpt": 5.7
        },
        "multimodalGrounded": {
          "aaMmmuPro": 74.2,
          "designArenaWebsite": 1203
        },
        "knowledge": {
          "artificialAnalysis": 35.31,
          "aaGpqaDiamond": 85.4,
          "aaHle": 28.5,
          "aaOmniscienceIndex": -8.7,
          "omniscienceAccuracy": 40.3,
          "omniscienceHallucinationRate": 82.2
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 73.1
        },
        "math": {
          "aaMath500": 99.4
        }
      }
    },
    {
      "slug": "glm-5v-turbo",
      "canonicalModelKey": "glm-5v-turbo",
      "model": "GLM-5V-Turbo",
      "creator": "Z.AI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": 62.09,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 50.56,
        "upper": 73.63
      },
      "rankingEligible": true,
      "overallRank": 53,
      "url": "https://benchlm.ai/models/glm-5v-turbo",
      "markdownUrl": "https://benchlm.ai/md/models/glm-5v-turbo.md",
      "id": 102,
      "releaseDate": "2026-03-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "glm-5",
        "familyName": "GLM-5",
        "variantType": "vision-turbo",
        "snapshotLabel": null,
        "baseFamilyModelKey": "glm-5",
        "relatedModelKeys": [
          "glm-5",
          "glm-5-reasoning",
          "glm-5-1",
          "glm-5-turbo"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 53,
        "categoryRanks": {
          "coding": 43
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "clawEval": 53.8,
          "tau2Bench": 98.5,
          "gertLabs": 30.76
        },
        "coding": {
          "aaSciCode": 43.5
        },
        "reasoning": {
          "lcr": 65.7,
          "critpt": 0.6
        },
        "multimodalGrounded": {
          "aaMmmuPro": 72.8,
          "designArenaWebsite": 1249
        },
        "knowledge": {
          "artificialAnalysis": 35.35,
          "aaGpqaDiamond": 80.9,
          "aaHle": 17.1,
          "aaOmniscienceIndex": -19.3,
          "omniscienceAccuracy": 29.3,
          "omniscienceHallucinationRate": 68.8
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 61.1
        },
        "math": {}
      }
    },
    {
      "slug": "mimo-v2-omni",
      "canonicalModelKey": "mimo-v2-omni",
      "model": "MiMo-V2-Omni",
      "creator": "Xiaomi",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": 62.17,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 51.23,
        "upper": 73.1
      },
      "rankingEligible": true,
      "overallRank": 51,
      "url": "https://benchlm.ai/models/mimo-v2-omni",
      "markdownUrl": "https://benchlm.ai/md/models/mimo-v2-omni.md",
      "id": 24,
      "releaseDate": "2026-03-18",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "mimo-v2-omni",
        "familyName": "MiMo-V2-Omni",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "mimo-v2-omni",
        "relatedModelKeys": [
          "mimo-v2-flash",
          "mimo-v2-pro"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 51,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "clawEval": 45.2,
          "tau2Bench": 91.2
        },
        "coding": {
          "sweVerified": 74.8,
          "aaSciCode": 36.7
        },
        "reasoning": {
          "lcr": 71.7,
          "critpt": 1.1
        },
        "multimodalGrounded": {
          "aaMmmuPro": 69.9
        },
        "knowledge": {
          "artificialAnalysis": 35.87,
          "aaGpqaDiamond": 82.8,
          "aaHle": 22.1,
          "aaOmniscienceIndex": -20.1,
          "omniscienceAccuracy": 19.3,
          "omniscienceHallucinationRate": 48.9
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 53.5
        },
        "math": {}
      }
    },
    {
      "slug": "grok-4-1-fast-reasoning",
      "canonicalModelKey": "grok-4-1-fast-reasoning",
      "model": "Grok 4.1 Fast (Reasoning)",
      "creator": "xAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "2M",
      "contextWindowTokens": 2000000,
      "displayScore": 59.55,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 45.62,
        "upper": 73.47
      },
      "rankingEligible": true,
      "overallRank": 76,
      "url": "https://benchlm.ai/models/grok-4-1-fast-reasoning",
      "markdownUrl": "https://benchlm.ai/md/models/grok-4-1-fast-reasoning.md",
      "id": 235,
      "releaseDate": "2025-11-19",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "grok-4-1-fast",
        "familyName": "Grok 4.1 Fast",
        "variantType": "reasoning",
        "snapshotLabel": "reasoning",
        "baseFamilyModelKey": "grok-4-1-fast",
        "relatedModelKeys": [
          "grok-4-1-fast"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 76,
        "categoryRanks": {
          "coding": 77
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 93.3
        },
        "coding": {
          "vibeCodeBench": 1.2,
          "aaSciCode": 44.2
        },
        "reasoning": {
          "lcr": 70.3,
          "critpt": 2.9
        },
        "multimodalGrounded": {
          "aaMmmuPro": 63.3
        },
        "knowledge": {
          "artificialAnalysis": 31.32,
          "aaGpqaDiamond": 85.3,
          "aaHle": 19.3,
          "aaOmniscienceIndex": -29.9,
          "omniscienceAccuracy": 25.1,
          "omniscienceHallucinationRate": 73.4
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 52.7
        },
        "math": {}
      }
    },
    {
      "slug": "deepseek-v3-2-thinking",
      "canonicalModelKey": "deepseek-v3-2-thinking",
      "model": "DeepSeek V3.2 (Thinking)",
      "creator": "DeepSeek",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 58.89,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 47.37,
        "upper": 70.4
      },
      "rankingEligible": true,
      "overallRank": 82,
      "url": "https://benchlm.ai/models/deepseek-v3-2-thinking",
      "markdownUrl": "https://benchlm.ai/md/models/deepseek-v3-2-thinking.md",
      "id": 72,
      "releaseDate": "2025-12-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "deepseek-v3-2",
        "familyName": "DeepSeek V3.2",
        "variantType": "reasoning",
        "snapshotLabel": null,
        "baseFamilyModelKey": "deepseek-v3-2",
        "relatedModelKeys": [
          "deepseek-v3-2"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 82,
        "categoryRanks": {
          "coding": 72
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {
          "vibeCodeBench": 5.108
        },
        "reasoning": {},
        "multimodalGrounded": {
          "designArenaWebsite": 1193
        },
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "grok-4-1-fast",
      "canonicalModelKey": "grok-4-1-fast",
      "model": "Grok 4.1 Fast",
      "creator": "xAI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 50.66,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 28.09,
        "upper": 73.22
      },
      "rankingEligible": true,
      "overallRank": 131,
      "url": "https://benchlm.ai/models/grok-4-1-fast",
      "markdownUrl": "https://benchlm.ai/md/models/grok-4-1-fast.md",
      "id": 21,
      "releaseDate": "2025-11-19",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "grok-4-1-fast",
        "familyName": "Grok 4.1 Fast",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "grok-4-1-fast",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 131,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 63.7,
          "gertLabs": 47.32
        },
        "coding": {
          "aaSciCode": 29.6
        },
        "reasoning": {
          "lcr": 28.7,
          "critpt": 0
        },
        "multimodalGrounded": {
          "aaMmmuPro": 48.4
        },
        "knowledge": {
          "artificialAnalysis": 17.03,
          "aaGpqaDiamond": 63.7,
          "aaHle": 5.1,
          "aaOmniscienceIndex": -50.9,
          "omniscienceAccuracy": 17.2,
          "omniscienceHallucinationRate": 82.3
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 36.5
        },
        "math": {}
      }
    },
    {
      "slug": "deepseek-v3-1-reasoning",
      "canonicalModelKey": "deepseek-v3-1-reasoning",
      "model": "DeepSeek V3.1 (Reasoning)",
      "creator": "DeepSeek",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 52.74,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 33.02,
        "upper": 72.46
      },
      "rankingEligible": true,
      "overallRank": 118,
      "url": "https://benchlm.ai/models/deepseek-v3-1-reasoning",
      "markdownUrl": "https://benchlm.ai/md/models/deepseek-v3-1-reasoning.md",
      "id": 155,
      "releaseDate": "2025-08-21",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "deepseek-v3-1",
        "familyName": "DeepSeek V3.1",
        "variantType": "reasoning",
        "snapshotLabel": null,
        "baseFamilyModelKey": "deepseek-v3-1",
        "relatedModelKeys": [
          "deepseek-v3-1"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 118,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 37.4
        },
        "coding": {
          "aaSciCode": 39.1
        },
        "reasoning": {
          "lcr": 56.7,
          "critpt": 2
        },
        "multimodalGrounded": {
          "designArenaWebsite": 1141
        },
        "knowledge": {
          "artificialAnalysis": 20.97,
          "aaGpqaDiamond": 77.9,
          "aaHle": 14.3,
          "aaOmniscienceIndex": -29.6,
          "omniscienceAccuracy": 29,
          "omniscienceHallucinationRate": 82.5
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 41.5
        },
        "math": {}
      }
    },
    {
      "slug": "deepseek-v3-1",
      "canonicalModelKey": "deepseek-v3-1",
      "model": "DeepSeek V3.1",
      "creator": "DeepSeek",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 52.92,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 33.51,
        "upper": 72.32
      },
      "rankingEligible": true,
      "overallRank": 116,
      "url": "https://benchlm.ai/models/deepseek-v3-1",
      "markdownUrl": "https://benchlm.ai/md/models/deepseek-v3-1.md",
      "id": 166,
      "releaseDate": "2025-08-21",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "deepseek-v3-1",
        "familyName": "DeepSeek V3.1",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "deepseek-v3-1",
        "relatedModelKeys": [
          "deepseek-v3-1-reasoning"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 116,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 34.8
        },
        "coding": {
          "aaSciCode": 36.7
        },
        "reasoning": {
          "lcr": 46.7,
          "critpt": 0
        },
        "multimodalGrounded": {
          "designArenaWebsite": 1141
        },
        "knowledge": {
          "artificialAnalysis": 21.37,
          "aaGpqaDiamond": 73.5,
          "aaHle": 6.7,
          "aaOmniscienceIndex": -42.7,
          "omniscienceAccuracy": 23.1,
          "omniscienceHallucinationRate": 85.7
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 37.8
        },
        "math": {}
      }
    },
    {
      "slug": "mistral-large-3",
      "canonicalModelKey": "mistral-large-3",
      "model": "Mistral Large 3",
      "creator": "Mistral",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 49.6,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 26.88,
        "upper": 72.32
      },
      "rankingEligible": true,
      "overallRank": 139,
      "url": "https://benchlm.ai/models/mistral-large-3",
      "markdownUrl": "https://benchlm.ai/md/models/mistral-large-3.md",
      "id": 99,
      "releaseDate": "2025-12-02",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "mistral-large-3",
        "familyName": "Mistral Large 3",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "mistral-large-3",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 139,
        "categoryRanks": {
          "agentic": 100,
          "coding": 142
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "aaAgenticIndex": 5.52,
          "tau2Bench": 24.6,
          "gdpvalAaNormalized": 7.2,
          "gdpvalAa": 644
        },
        "coding": {
          "aaCodingIndex": 20.07,
          "aaSciCode": 36.2
        },
        "reasoning": {
          "lcr": 34.7,
          "critpt": 0
        },
        "multimodalGrounded": {
          "aaMmmuPro": 55.7
        },
        "knowledge": {
          "artificialAnalysis": 15.92,
          "aaGpqaDiamond": 68,
          "aaHle": 4.2,
          "aaOmniscienceIndex": -39.6,
          "omniscienceAccuracy": 25,
          "omniscienceHallucinationRate": 86
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 36.2
        },
        "math": {}
      }
    },
    {
      "slug": "glm-4-5",
      "canonicalModelKey": "glm-4-5",
      "model": "GLM-4.5",
      "creator": "Z.AI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 58.32,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 46.81,
        "upper": 69.84
      },
      "rankingEligible": true,
      "overallRank": 86,
      "url": "https://benchlm.ai/models/glm-4-5",
      "markdownUrl": "https://benchlm.ai/md/models/glm-4-5.md",
      "id": 164,
      "releaseDate": "2025-06-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "glm-4-5",
        "familyName": "GLM-4.5",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "glm-4-5",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 86,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {
          "designArenaWebsite": 1188
        },
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "grok-4-fast-reasoning",
      "canonicalModelKey": "grok-4-fast-reasoning",
      "model": "Grok 4 Fast (Reasoning)",
      "creator": "xAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "2M",
      "contextWindowTokens": 2000000,
      "displayScore": 55.76,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 41.74,
        "upper": 69.78
      },
      "rankingEligible": true,
      "overallRank": 101,
      "url": "https://benchlm.ai/models/grok-4-fast-reasoning",
      "markdownUrl": "https://benchlm.ai/md/models/grok-4-fast-reasoning.md",
      "id": 236,
      "releaseDate": "2025-09-19",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "grok-4-fast",
        "familyName": "Grok 4 Fast",
        "variantType": "reasoning",
        "snapshotLabel": "reasoning",
        "baseFamilyModelKey": "grok-4-fast-reasoning",
        "relatedModelKeys": [
          "grok-4"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 101,
        "categoryRanks": {
          "coding": 111
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 65.8
        },
        "coding": {
          "vibeCodeBench": 0,
          "aaSciCode": 44.2
        },
        "reasoning": {
          "lcr": 71,
          "critpt": 2.9
        },
        "multimodalGrounded": {
          "aaMmmuPro": 61.8
        },
        "knowledge": {
          "artificialAnalysis": 27.95,
          "aaGpqaDiamond": 84.7,
          "aaHle": 19.1,
          "aaOmniscienceIndex": -29.9,
          "omniscienceAccuracy": 22.8,
          "omniscienceHallucinationRate": 68.3
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 50.5
        },
        "math": {}
      }
    },
    {
      "slug": "deepseek-r1",
      "canonicalModelKey": "deepseek-r1",
      "model": "DeepSeek-R1",
      "creator": "DeepSeek",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 50.96,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 32.5,
        "upper": 69.43
      },
      "rankingEligible": true,
      "overallRank": 124,
      "url": "https://benchlm.ai/models/deepseek-r1",
      "markdownUrl": "https://benchlm.ai/md/models/deepseek-r1.md",
      "id": 142,
      "releaseDate": "2025-01-20",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "deepseek-r1",
        "familyName": "DeepSeek-R1",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "deepseek-r1",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 124,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 36.5
        },
        "coding": {
          "aaSciCode": 40.3
        },
        "reasoning": {
          "lcr": 56.7,
          "critpt": 1.4
        },
        "multimodalGrounded": {},
        "knowledge": {
          "artificialAnalysis": 20.37,
          "aaGpqaDiamond": 81.3,
          "aaHle": 15.8,
          "aaOmniscienceIndex": -27.4,
          "omniscienceAccuracy": 30.5,
          "omniscienceHallucinationRate": 83.4
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 39.6
        },
        "math": {}
      }
    },
    {
      "slug": "gpt-5-3-codex-spark",
      "canonicalModelKey": "gpt-5-3-codex-spark",
      "model": "GPT-5.3-Codex-Spark",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": 57.65,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 46.14,
        "upper": 69.17
      },
      "rankingEligible": true,
      "overallRank": 90,
      "url": "https://benchlm.ai/models/gpt-5-3-codex-spark",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-5-3-codex-spark.md",
      "id": 180,
      "releaseDate": "2026-02-12",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-5-3-codex",
        "familyName": "GPT-5.3 Codex",
        "variantType": "spark",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-5-3-codex",
        "relatedModelKeys": [
          "gpt-5-3-codex"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 90,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 15,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "step-3-5-flash",
      "canonicalModelKey": "step-3-5-flash",
      "model": "Step 3.5 Flash",
      "creator": "StepFun",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": 54.22,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 40.09,
        "upper": 68.34
      },
      "rankingEligible": true,
      "overallRank": 108,
      "url": "https://benchlm.ai/models/step-3-5-flash",
      "markdownUrl": "https://benchlm.ai/md/models/step-3-5-flash.md",
      "id": 181,
      "releaseDate": "2026-01-20",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "step-3-5-flash",
        "familyName": "Step 3.5 Flash",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "step-3-5-flash",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 108,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 23,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "minimax-m2-5",
      "canonicalModelKey": "minimax-m2-5",
      "model": "MiniMax M2.5",
      "creator": "MiniMax",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 58.49,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 49.9,
        "upper": 67.09
      },
      "rankingEligible": true,
      "overallRank": 85,
      "url": "https://benchlm.ai/models/minimax-m2-5",
      "markdownUrl": "https://benchlm.ai/md/models/minimax-m2-5.md",
      "id": 189,
      "releaseDate": "2025-10-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "minimax-m2-5",
        "familyName": "MiniMax M2.5",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "minimax-m2-5",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 85,
        "categoryRanks": {
          "agentic": 97,
          "coding": 68
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 38,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {
          "vibeCodeBench": 14.852
        },
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "o1-preview",
      "canonicalModelKey": "o1-preview",
      "model": "o1-preview",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": 48.48,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 28.75,
        "upper": 68.21
      },
      "rankingEligible": true,
      "overallRank": 143,
      "url": "https://benchlm.ai/models/o1-preview",
      "markdownUrl": "https://benchlm.ai/md/models/o1-preview.md",
      "id": 20,
      "releaseDate": "2024-09-12",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "o1",
        "familyName": "o1",
        "variantType": "snapshot",
        "snapshotLabel": "preview",
        "baseFamilyModelKey": "o1",
        "relatedModelKeys": [
          "o1",
          "o1-pro"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 143,
        "categoryRanks": {
          "coding": 98
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {
          "aaCodingIndex": 34.05
        },
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {
          "artificialAnalysis": 17.21,
          "aaGpqaDiamond": 76.5
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "trinity-large-preview",
      "canonicalModelKey": "trinity-large-preview",
      "model": "Trinity-Large-Preview",
      "creator": "Arcee AI",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "512K",
      "contextWindowTokens": 512000,
      "displayScore": 56.67,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 45.15,
        "upper": 68.18
      },
      "rankingEligible": true,
      "overallRank": 97,
      "url": "https://benchlm.ai/models/trinity-large-preview",
      "markdownUrl": "https://benchlm.ai/md/models/trinity-large-preview.md",
      "id": 175,
      "releaseDate": "2026-01-27",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "trinity-large",
        "familyName": "Trinity Large",
        "variantType": "preview",
        "snapshotLabel": "preview",
        "baseFamilyModelKey": "trinity-large-thinking",
        "relatedModelKeys": [
          "trinity-large-thinking"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 97,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 4,
        "verifiedBenchmarkCount": 4,
        "rankableBenchmarkCount": 4,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 90.1,
          "gdpvalAaNormalized": 3.3,
          "gdpvalAa": 565,
          "aaAgenticIndex": 3.72
        },
        "coding": {
          "aaSciCode": 36.1,
          "aaCodingIndex": 25.77
        },
        "reasoning": {
          "lcr": 38.3,
          "critpt": 0.9
        },
        "multimodalGrounded": {
          "designArenaWebsite": 1155
        },
        "knowledge": {
          "mmlu": 87.2,
          "mmluProArcee": 75.2,
          "gpqaDiamond": 63.3,
          "artificialAnalysis": 18.66,
          "aaGpqaDiamond": 75.2,
          "aaHle": 15.8,
          "aaOmniscienceIndex": -44.1,
          "omniscienceAccuracy": 22.5,
          "omniscienceHallucinationRate": 85.9
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 56.3
        },
        "math": {
          "aime2025Arcee": 24
        }
      }
    },
    {
      "slug": "glm-4-5-air",
      "canonicalModelKey": "glm-4-5-air",
      "model": "GLM-4.5-Air",
      "creator": "Z.AI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 47.08,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 28.25,
        "upper": 65.92
      },
      "rankingEligible": true,
      "overallRank": 156,
      "url": "https://benchlm.ai/models/glm-4-5-air",
      "markdownUrl": "https://benchlm.ai/md/models/glm-4-5-air.md",
      "id": 163,
      "releaseDate": "2025-06-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "glm-4-5-air",
        "familyName": "GLM-4.5-Air",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "glm-4-5-air",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 156,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 46.5
        },
        "coding": {
          "aaSciCode": 30.6
        },
        "reasoning": {
          "lcr": 45.7,
          "critpt": 0
        },
        "multimodalGrounded": {
          "designArenaWebsite": 1165
        },
        "knowledge": {
          "artificialAnalysis": 16.66,
          "aaGpqaDiamond": 73.3,
          "aaHle": 7,
          "aaOmniscienceIndex": -61.5,
          "omniscienceAccuracy": 16.3,
          "omniscienceHallucinationRate": 92.9
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 37.6
        },
        "math": {}
      }
    },
    {
      "slug": "trinity-large-thinking",
      "canonicalModelKey": "trinity-large-thinking",
      "model": "Trinity-Large-Thinking",
      "creator": "Arcee AI",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "512K",
      "contextWindowTokens": 512000,
      "displayScore": 47.92,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 30.73,
        "upper": 65.11
      },
      "rankingEligible": true,
      "overallRank": 148,
      "url": "https://benchlm.ai/models/trinity-large-thinking",
      "markdownUrl": "https://benchlm.ai/md/models/trinity-large-thinking.md",
      "id": 116,
      "releaseDate": "2026-03-10",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "trinity-large",
        "familyName": "Trinity Large",
        "variantType": "thinking",
        "snapshotLabel": null,
        "baseFamilyModelKey": "trinity-large-thinking",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 148,
        "categoryRanks": {
          "agentic": 107,
          "coding": 139
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 4,
        "verifiedBenchmarkCount": 4,
        "rankableBenchmarkCount": 4,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 90.1,
          "gertLabs": 32.55,
          "gdpvalAaNormalized": 3.3,
          "gdpvalAa": 565,
          "aaAgenticIndex": 3.72
        },
        "coding": {
          "sweVerifiedArcee": 63.2,
          "aaSciCode": 36.1,
          "aaCodingIndex": 25.77
        },
        "reasoning": {
          "lcr": 38.3,
          "critpt": 0.9
        },
        "multimodalGrounded": {
          "designArenaWebsite": 1155
        },
        "knowledge": {
          "gpqaDiamond": 76.3,
          "mmluProArcee": 83.4,
          "artificialAnalysis": 18.66,
          "aaGpqaDiamond": 75.2,
          "aaHle": 15.8,
          "aaOmniscienceIndex": -44.1,
          "omniscienceAccuracy": 22.5,
          "omniscienceHallucinationRate": 85.9
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 56.3
        },
        "math": {
          "aime2025Arcee": 96.3
        }
      }
    },
    {
      "slug": "glm-4-7-flash",
      "canonicalModelKey": "glm-4-7-flash",
      "model": "GLM-4.7-Flash",
      "creator": "Z.AI",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": 50.35,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 36.42,
        "upper": 64.28
      },
      "rankingEligible": true,
      "overallRank": 137,
      "url": "https://benchlm.ai/models/glm-4-7-flash",
      "markdownUrl": "https://benchlm.ai/md/models/glm-4-7-flash.md",
      "id": 186,
      "releaseDate": "2025-10-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "glm-4-7-flash",
        "familyName": "GLM-4.7-Flash",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "glm-4-7-flash",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 137,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 35,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "gemma-3-27b",
      "canonicalModelKey": "gemma-3-27b",
      "model": "Gemma 3 27B",
      "creator": "Google",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": 41.26,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 16.97,
        "upper": 65.54
      },
      "rankingEligible": true,
      "overallRank": 189,
      "url": "https://benchlm.ai/models/gemma-3-27b",
      "markdownUrl": "https://benchlm.ai/md/models/gemma-3-27b.md",
      "id": 165,
      "releaseDate": "2025-03-12",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gemma-3-27b",
        "familyName": "Gemma 3 27B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gemma-3-27b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 189,
        "categoryRanks": {
          "agentic": 125,
          "coding": 130
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "aaAgenticIndex": 0.27,
          "tau2Bench": 10.5,
          "gdpvalAaNormalized": 0,
          "gdpvalAa": -119
        },
        "coding": {
          "aaCodingIndex": 10.06,
          "aaSciCode": 21.2
        },
        "reasoning": {
          "lcr": 6.3,
          "critpt": 0
        },
        "multimodalGrounded": {
          "aaMmmuPro": 48
        },
        "knowledge": {
          "artificialAnalysis": 7.36,
          "aaGpqaDiamond": 42.8,
          "aaHle": 4.4,
          "aaOmniscienceIndex": -67.2,
          "omniscienceAccuracy": 13,
          "omniscienceHallucinationRate": 92.1
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 31.8
        },
        "math": {}
      }
    },
    {
      "slug": "mistral-small-4",
      "canonicalModelKey": "mistral-small-4",
      "model": "Mistral Small 4",
      "creator": "Mistral",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": 46.35,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 34.84,
        "upper": 57.87
      },
      "rankingEligible": true,
      "overallRank": 160,
      "url": "https://benchlm.ai/models/mistral-small-4",
      "markdownUrl": "https://benchlm.ai/md/models/mistral-small-4.md",
      "id": 66,
      "releaseDate": "2026-02-20",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "mistral-small-4",
        "familyName": "Mistral Small 4",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "mistral-small-4",
        "relatedModelKeys": [
          "mistral-small-4-reasoning"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 160,
        "categoryRanks": {
          "agentic": 99,
          "coding": 102
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "aaAgenticIndex": 4.63,
          "tau2Bench": 41.2,
          "gdpvalAaNormalized": 4.5,
          "gdpvalAa": 590
        },
        "coding": {
          "aaCodingIndex": 26.64,
          "aaSciCode": 38
        },
        "reasoning": {
          "lcr": 47.3,
          "critpt": 0.3
        },
        "multimodalGrounded": {
          "aaMmmuPro": 56.8
        },
        "knowledge": {
          "artificialAnalysis": 19.71,
          "aaGpqaDiamond": 76.9,
          "aaHle": 9.9,
          "aaOmniscienceIndex": -30.4,
          "omniscienceAccuracy": 21.7,
          "omniscienceHallucinationRate": 66.5
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 48.2
        },
        "math": {}
      }
    },
    {
      "slug": "gpt-oss-120b",
      "canonicalModelKey": "gpt-oss-120b",
      "model": "GPT-OSS 120B",
      "creator": "OpenAI",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 49.22,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 36.05,
        "upper": 62.4
      },
      "rankingEligible": true,
      "overallRank": 141,
      "url": "https://benchlm.ai/models/gpt-oss-120b",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-oss-120b.md",
      "id": 129,
      "releaseDate": "2025-08-05",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-oss",
        "familyName": "GPT-OSS",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-oss-120b",
        "relatedModelKeys": [
          "gpt-oss-20b"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": 56.4,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 60.8,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 141,
        "categoryRanks": {
          "agentic": 91,
          "coding": 101,
          "knowledge": 40
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "aaAgenticIndex": 13.43,
          "apexAgentsAa": 3.1,
          "tau2Bench": 65.8,
          "gdpvalAaNormalized": 15.1,
          "gdpvalAa": 803,
          "gertLabs": 29.61,
          "aaItbench": 5.6,
          "aaHarveyLab": 13.9,
          "terminalBenchHard": 23.5
        },
        "coding": {
          "reactNativeEvals": 71.6,
          "aaCodingIndex": 30.44,
          "aaSciCode": 38.9,
          "aaLiveCodeBench": 87.8
        },
        "reasoning": {
          "lcr": 51,
          "critpt": 1.1
        },
        "multimodalGrounded": {
          "designArenaWebsite": 986
        },
        "knowledge": {
          "artificialAnalysis": 24.13,
          "aaGpqaDiamond": 78.2,
          "aaHle": 19.6,
          "aaOmniscienceIndex": -49.2,
          "omniscienceAccuracy": 21.8,
          "omniscienceHallucinationRate": 90.8,
          "aaOpennessIndex": 38.9,
          "aaMmluPro": 80.8
        },
        "multilingual": {
          "aaGlobalMmluLite": 82.8
        },
        "instructionFollowing": {
          "aaIfBench": 69
        },
        "math": {
          "aaAime2025": 93.4
        }
      }
    },
    {
      "slug": "deepseek-llm-2-0",
      "canonicalModelKey": "deepseek-llm-2-0",
      "model": "DeepSeek LLM 2.0",
      "creator": "DeepSeek",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 55.24,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 43.72,
        "upper": 66.75
      },
      "rankingEligible": true,
      "overallRank": 103,
      "url": "https://benchlm.ai/models/deepseek-llm-2-0",
      "markdownUrl": "https://benchlm.ai/md/models/deepseek-llm-2-0.md",
      "id": 79,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "deepseek-llm-2-0",
        "familyName": "DeepSeek LLM 2.0",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "deepseek-llm-2-0",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 103,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "gpt-5-1-codex-max",
      "canonicalModelKey": "gpt-5-1-codex-max",
      "model": "GPT-5.1-Codex-Max",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "400K",
      "contextWindowTokens": 400000,
      "displayScore": 55.19,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 43.67,
        "upper": 66.7
      },
      "rankingEligible": true,
      "overallRank": 104,
      "url": "https://benchlm.ai/models/gpt-5-1-codex-max",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-5-1-codex-max.md",
      "id": 1,
      "releaseDate": "2025-11-19",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-5-1-codex-max",
        "familyName": "GPT-5.1-Codex-Max",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-5-1-codex-max",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 104,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 83
        },
        "coding": {
          "vibeCodeBench": 22.168,
          "aaSciCode": 40.2
        },
        "reasoning": {
          "lcr": 69,
          "critpt": 5.7
        },
        "multimodalGrounded": {
          "aaMmmuPro": 72.5
        },
        "knowledge": {
          "artificialAnalysis": 35.6,
          "aaGpqaDiamond": 86,
          "aaHle": 25.7,
          "aaOmniscienceIndex": -6.5,
          "omniscienceAccuracy": 39.9,
          "omniscienceHallucinationRate": 77.2
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 70
        },
        "math": {}
      }
    },
    {
      "slug": "mercury-2",
      "canonicalModelKey": "mercury-2",
      "model": "Mercury 2",
      "creator": "Inception",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 48.13,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 34.94,
        "upper": 61.33
      },
      "rankingEligible": true,
      "overallRank": 147,
      "url": "https://benchlm.ai/models/mercury-2",
      "markdownUrl": "https://benchlm.ai/md/models/mercury-2.md",
      "id": 184,
      "releaseDate": "2026-02-24",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "mercury-2",
        "familyName": "Mercury 2",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "mercury-2",
        "relatedModelKeys": [
          "mercury-2-5-preview"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 147,
        "categoryRanks": {
          "agentic": 101,
          "coding": 143
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 29,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "gpt-5-2-codex",
      "canonicalModelKey": "gpt-5-2-codex",
      "model": "GPT-5.2-Codex",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "400K",
      "contextWindowTokens": 400000,
      "displayScore": 57.86,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 54.73,
        "upper": 61
      },
      "rankingEligible": true,
      "overallRank": 88,
      "url": "https://benchlm.ai/models/gpt-5-2-codex",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-5-2-codex.md",
      "id": 4,
      "releaseDate": "2025-12-18",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-5-2-codex",
        "familyName": "GPT-5.2-Codex",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-5-2-codex",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": 58.6,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 88,
        "categoryRanks": {
          "agentic": 50,
          "coding": 47
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 2,
        "verifiedBenchmarkCount": 2,
        "rankableBenchmarkCount": 2,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 92.1,
          "gertLabs": 51.79,
          "jobBench": 26
        },
        "coding": {
          "vibeCodeBench": 37.912,
          "aaSciCode": 54.6
        },
        "reasoning": {
          "lcr": 79.3,
          "critpt": 8.7
        },
        "multimodalGrounded": {
          "aaMmmuPro": 76.3
        },
        "knowledge": {
          "artificialAnalysis": 41.22,
          "aaGpqaDiamond": 89.9,
          "aaHle": 35.7,
          "aaOmniscienceIndex": -2.2,
          "omniscienceAccuracy": 41.1,
          "omniscienceHallucinationRate": 73.4
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 77.6
        },
        "math": {}
      }
    },
    {
      "slug": "gpt-5-medium",
      "canonicalModelKey": "gpt-5-medium",
      "model": "GPT-5 (medium)",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 54.06,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 49.86,
        "upper": 58.27
      },
      "rankingEligible": true,
      "overallRank": 109,
      "url": "https://benchlm.ai/models/gpt-5-medium",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-5-medium.md",
      "id": 16,
      "releaseDate": "2025-08-07",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-5",
        "familyName": "GPT-5",
        "variantType": "reasoning",
        "snapshotLabel": "medium",
        "baseFamilyModelKey": "gpt-5-high",
        "relatedModelKeys": [
          "gpt-5-high",
          "gpt-5-mini",
          "gpt-5-nano"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 109,
        "categoryRanks": {
          "coding": 65
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 86.5
        },
        "coding": {
          "aaSciCode": 41.1
        },
        "reasoning": {
          "lcr": 76.3,
          "critpt": 0
        },
        "multimodalGrounded": {
          "aaMmmuPro": 74.3,
          "designArenaWebsite": 1203
        },
        "knowledge": {
          "artificialAnalysis": 34.57,
          "aaGpqaDiamond": 84.2,
          "aaHle": 25.4,
          "aaOmniscienceIndex": -10.9,
          "omniscienceAccuracy": 39.5,
          "omniscienceHallucinationRate": 83.2
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 70.6
        },
        "math": {
          "aaMath500": 99.1
        }
      }
    },
    {
      "slug": "claude-3-opus",
      "canonicalModelKey": "claude-3-opus",
      "model": "Claude 3 Opus",
      "creator": "Anthropic",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": 40.57,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 22.75,
        "upper": 58.38
      },
      "rankingEligible": true,
      "overallRank": 193,
      "url": "https://benchlm.ai/models/claude-3-opus",
      "markdownUrl": "https://benchlm.ai/md/models/claude-3-opus.md",
      "id": 124,
      "releaseDate": "2024-03-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "claude-3-opus",
        "familyName": "Claude 3 Opus",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "claude-3-opus",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 193,
        "categoryRanks": {
          "coding": 129
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {
          "aaCodingIndex": 19.53,
          "aaSciCode": 23.3
        },
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {
          "artificialAnalysis": 11.76,
          "aaGpqaDiamond": 48.9,
          "aaHle": 2.8
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "nemotron-3-nano-30b",
      "canonicalModelKey": "nemotron-3-nano-30b",
      "model": "Nemotron 3 Nano 30B",
      "creator": "NVIDIA",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": 53.63,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 42.12,
        "upper": 65.15
      },
      "rankingEligible": true,
      "overallRank": 111,
      "url": "https://benchlm.ai/models/nemotron-3-nano-30b",
      "markdownUrl": "https://benchlm.ai/md/models/nemotron-3-nano-30b.md",
      "id": 145,
      "releaseDate": "2026-01-15",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "nemotron-3-nano-30b",
        "familyName": "Nemotron 3 Nano 30B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "nemotron-3-nano-30b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 111,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 40.9,
          "gdpvalAaNormalized": 0,
          "gdpvalAa": 493,
          "aaAgenticIndex": 1.99
        },
        "coding": {
          "aaSciCode": 29.6,
          "aaCodingIndex": 14.37
        },
        "reasoning": {
          "lcr": 37.3,
          "critpt": 0.9
        },
        "multimodalGrounded": {},
        "knowledge": {
          "artificialAnalysis": 14.54,
          "aaGpqaDiamond": 75.7,
          "aaHle": 11.4,
          "aaOmniscienceIndex": -51.6,
          "omniscienceAccuracy": 17.3,
          "omniscienceHallucinationRate": 83.3
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 71.1
        },
        "math": {}
      }
    },
    {
      "slug": "gpt-oss-20b",
      "canonicalModelKey": "gpt-oss-20b",
      "model": "GPT-OSS 20B",
      "creator": "OpenAI",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 42.25,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 27.12,
        "upper": 57.39
      },
      "rankingEligible": true,
      "overallRank": 185,
      "url": "https://benchlm.ai/models/gpt-oss-20b",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-oss-20b.md",
      "id": 167,
      "releaseDate": "2025-08-05",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-oss",
        "familyName": "GPT-OSS",
        "variantType": "mini",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-oss-120b",
        "relatedModelKeys": [
          "gpt-oss-120b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 185,
        "categoryRanks": {
          "agentic": 123,
          "coding": 127
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "aaAgenticIndex": 3.1,
          "apexAgentsAa": 0.7,
          "tau2Bench": 60.2,
          "gdpvalAaNormalized": 3.4,
          "gdpvalAa": 569
        },
        "coding": {
          "reactNativeEvals": 71,
          "aaCodingIndex": 20.7,
          "aaSciCode": 34.4
        },
        "reasoning": {
          "lcr": 33.3,
          "critpt": 1.4
        },
        "multimodalGrounded": {
          "designArenaWebsite": 871
        },
        "knowledge": {
          "artificialAnalysis": 15.23,
          "aaGpqaDiamond": 68.8,
          "aaHle": 11,
          "aaOmniscienceIndex": -63,
          "omniscienceAccuracy": 16,
          "omniscienceHallucinationRate": 94.1
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 65.1
        },
        "math": {}
      }
    },
    {
      "slug": "gpt-4o-mini",
      "canonicalModelKey": "gpt-4o-mini",
      "model": "GPT-4o mini",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 37.43,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 16.67,
        "upper": 58.19
      },
      "rankingEligible": true,
      "overallRank": 204,
      "url": "https://benchlm.ai/models/gpt-4o-mini",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-4o-mini.md",
      "id": 101,
      "releaseDate": "2024-07-18",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-4o",
        "familyName": "GPT-4o",
        "variantType": "mini",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-4o",
        "relatedModelKeys": [
          "gpt-4o"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 204,
        "categoryRanks": {
          "agentic": 129,
          "coding": 135
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "aaAgenticIndex": 0.96,
          "gdpvalAaNormalized": 0,
          "gdpvalAa": 239
        },
        "coding": {
          "aaSciCode": 22.9,
          "aaCodingIndex": 11.38
        },
        "reasoning": {},
        "multimodalGrounded": {
          "aaMmmuPro": 41.5
        },
        "knowledge": {
          "artificialAnalysis": 6.67,
          "aaGpqaDiamond": 42.6,
          "aaHle": 4.2
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 31
        },
        "math": {},
        "korean": {}
      }
    },
    {
      "slug": "mistral-large-2",
      "canonicalModelKey": "mistral-large-2",
      "model": "Mistral Large 2",
      "creator": "Mistral",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 42.12,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 30.61,
        "upper": 53.63
      },
      "rankingEligible": true,
      "overallRank": 186,
      "url": "https://benchlm.ai/models/mistral-large-2",
      "markdownUrl": "https://benchlm.ai/md/models/mistral-large-2.md",
      "id": 100,
      "releaseDate": "2024-07-24",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "mistral-large-2",
        "familyName": "Mistral Large 2",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "mistral-large-2",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 186,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 30.7
        },
        "coding": {
          "aaSciCode": 29.2
        },
        "reasoning": {
          "lcr": 5.3,
          "critpt": 0
        },
        "multimodalGrounded": {},
        "knowledge": {
          "artificialAnalysis": 8.99,
          "aaGpqaDiamond": 48.6,
          "aaHle": 3.3,
          "aaOmniscienceIndex": -34.4,
          "omniscienceAccuracy": 19.9,
          "omniscienceHallucinationRate": 67.7
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 31.2
        },
        "math": {}
      }
    },
    {
      "slug": "qwen2-5-72b",
      "canonicalModelKey": "qwen2-5-72b",
      "model": "Qwen2.5-72B",
      "creator": "Alibaba",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 52.83,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 41.32,
        "upper": 64.35
      },
      "rankingEligible": true,
      "overallRank": 117,
      "url": "https://benchlm.ai/models/qwen2-5-72b",
      "markdownUrl": "https://benchlm.ai/md/models/qwen2-5-72b.md",
      "id": 73,
      "releaseDate": "2024-09-19",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "qwen2-5-72b",
        "familyName": "Qwen2.5-72B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "qwen2-5-72b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 117,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "llama-3-1-405b",
      "canonicalModelKey": "llama-3-1-405b",
      "model": "Llama 3.1 405B",
      "creator": "Meta",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 52.38,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 40.87,
        "upper": 63.9
      },
      "rankingEligible": true,
      "overallRank": 120,
      "url": "https://benchlm.ai/models/llama-3-1-405b",
      "markdownUrl": "https://benchlm.ai/md/models/llama-3-1-405b.md",
      "id": 97,
      "releaseDate": "2024-07-23",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "llama-3-1-405b",
        "familyName": "Llama 3.1 405B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "llama-3-1-405b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 120,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 19
        },
        "coding": {
          "aaSciCode": 29.9
        },
        "reasoning": {
          "lcr": 24.7,
          "critpt": 0
        },
        "multimodalGrounded": {},
        "knowledge": {
          "artificialAnalysis": 8.32,
          "aaGpqaDiamond": 51.5,
          "aaHle": 4,
          "aaOmniscienceIndex": -17.1,
          "omniscienceAccuracy": 23.2,
          "omniscienceHallucinationRate": 52.4
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 39
        },
        "math": {}
      }
    },
    {
      "slug": "nemotron-3-super-120b-a12b",
      "canonicalModelKey": "nemotron-3-super-120b-a12b",
      "model": "Nemotron 3 Super 120B A12B",
      "creator": "NVIDIA",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": 51.63,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 40.12,
        "upper": 63.15
      },
      "rankingEligible": true,
      "overallRank": 121,
      "url": "https://benchlm.ai/models/nemotron-3-super-120b-a12b",
      "markdownUrl": "https://benchlm.ai/md/models/nemotron-3-super-120b-a12b.md",
      "id": 187,
      "releaseDate": "2026-01-15",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "nemotron-3-super-120b-a12b",
        "familyName": "Nemotron 3 Super 120B A12B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "nemotron-3-super-120b-a12b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 121,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 19,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "llama-3-70b",
      "canonicalModelKey": "llama-3-70b",
      "model": "Llama 3 70B",
      "creator": "Meta",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 51.48,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 39.97,
        "upper": 63
      },
      "rankingEligible": true,
      "overallRank": 122,
      "url": "https://benchlm.ai/models/llama-3-70b",
      "markdownUrl": "https://benchlm.ai/md/models/llama-3-70b.md",
      "id": 143,
      "releaseDate": "2024-04-18",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "llama-3-70b",
        "familyName": "Llama 3 70B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "llama-3-70b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 122,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "qwen2-5-coder-32b-instruct",
      "canonicalModelKey": "qwen2-5-coder-32b-instruct",
      "model": "Qwen2.5 Coder 32B Instruct",
      "creator": "Alibaba",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 34.21,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 17.53,
        "upper": 50.89
      },
      "rankingEligible": true,
      "overallRank": 208,
      "url": "https://benchlm.ai/models/qwen2-5-coder-32b-instruct",
      "markdownUrl": "https://benchlm.ai/md/models/qwen2-5-coder-32b-instruct.md",
      "id": 216,
      "releaseDate": "2025-01-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "qwen2-5-coder",
        "familyName": "Qwen2.5 Coder",
        "variantType": "32b-instruct",
        "snapshotLabel": null,
        "baseFamilyModelKey": "qwen2-5-coder-32b-instruct",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 208,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {
          "aaSciCode": 27.1
        },
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {
          "artificialAnalysis": 6.88,
          "aaGpqaDiamond": 41.7,
          "aaHle": 3.5
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "deepseek-coder-2-0",
      "canonicalModelKey": "deepseek-coder-2-0",
      "model": "DeepSeek Coder 2.0",
      "creator": "DeepSeek",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 50.93,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 39.42,
        "upper": 62.45
      },
      "rankingEligible": true,
      "overallRank": 125,
      "url": "https://benchlm.ai/models/deepseek-coder-2-0",
      "markdownUrl": "https://benchlm.ai/md/models/deepseek-coder-2-0.md",
      "id": 61,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "deepseek-coder-2-0",
        "familyName": "DeepSeek Coder 2.0",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "deepseek-coder-2-0",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 125,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "seed-1-6",
      "canonicalModelKey": "seed-1-6",
      "model": "Seed 1.6",
      "creator": "ByteDance",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": 50.88,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 39.37,
        "upper": 62.4
      },
      "rankingEligible": true,
      "overallRank": 126,
      "url": "https://benchlm.ai/models/seed-1-6",
      "markdownUrl": "https://benchlm.ai/md/models/seed-1-6.md",
      "id": 183,
      "releaseDate": "2025-06-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "seed-1-6",
        "familyName": "Seed 1.6",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "seed-1-6",
        "relatedModelKeys": [
          "seed-1-6-flash"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 126,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 2,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "nemotron-3-super-100b",
      "canonicalModelKey": "nemotron-3-super-100b",
      "model": "Nemotron 3 Super 100B",
      "creator": "NVIDIA",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 50.73,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 39.22,
        "upper": 62.25
      },
      "rankingEligible": true,
      "overallRank": 130,
      "url": "https://benchlm.ai/models/nemotron-3-super-100b",
      "markdownUrl": "https://benchlm.ai/md/models/nemotron-3-super-100b.md",
      "id": 92,
      "releaseDate": "2026-01-15",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "nemotron-3-super-100b",
        "familyName": "Nemotron 3 Super 100B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "nemotron-3-super-100b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 130,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "clawEval": 5.5,
          "aaAgenticIndex": 8.79,
          "apexAgentsAa": 1.8,
          "tau2Bench": 67.8,
          "gdpvalAaNormalized": 10,
          "gdpvalAa": 701
        },
        "coding": {
          "aaCodingIndex": 37.72,
          "aaSciCode": 36
        },
        "reasoning": {
          "lcr": 60.3,
          "critpt": 3.1
        },
        "multimodalGrounded": {},
        "knowledge": {
          "artificialAnalysis": 25.67,
          "aaGpqaDiamond": 80,
          "aaHle": 20.8,
          "aaOmniscienceIndex": -41.5,
          "omniscienceAccuracy": 24.3,
          "omniscienceHallucinationRate": 87
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 71.5
        },
        "math": {}
      }
    },
    {
      "slug": "gemini-1-5-pro",
      "canonicalModelKey": "gemini-1-5-pro",
      "model": "Gemini 1.5 Pro",
      "creator": "Google",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "2M",
      "contextWindowTokens": 2000000,
      "displayScore": 35.14,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 21.08,
        "upper": 49.2
      },
      "rankingEligible": true,
      "overallRank": 207,
      "url": "https://benchlm.ai/models/gemini-1-5-pro",
      "markdownUrl": "https://benchlm.ai/md/models/gemini-1-5-pro.md",
      "id": 114,
      "releaseDate": "2024-02-15",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gemini-1-5-pro",
        "familyName": "Gemini 1.5 Pro",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gemini-1-5-pro",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 207,
        "categoryRanks": {
          "coding": 132
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {
          "aaCodingIndex": 23.63,
          "aaSciCode": 29.5
        },
        "reasoning": {},
        "multimodalGrounded": {
          "aaMmmuPro": 55
        },
        "knowledge": {
          "artificialAnalysis": 9.85,
          "aaGpqaDiamond": 58.9,
          "aaHle": 4.6
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "qwen2-5-1m",
      "canonicalModelKey": "qwen2-5-1m",
      "model": "Qwen2.5-1M",
      "creator": "Alibaba",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 50.54,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 39.02,
        "upper": 62.05
      },
      "rankingEligible": true,
      "overallRank": 134,
      "url": "https://benchlm.ai/models/qwen2-5-1m",
      "markdownUrl": "https://benchlm.ai/md/models/qwen2-5-1m.md",
      "id": 54,
      "releaseDate": "2025-01-27",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "qwen2-5-1m",
        "familyName": "Qwen2.5-1M",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "qwen2-5-1m",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 134,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "phi-4",
      "canonicalModelKey": "phi-4",
      "model": "Phi-4",
      "creator": "Microsoft",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "16K",
      "contextWindowTokens": 16000,
      "displayScore": 22.55,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 0,
        "upper": 45.61
      },
      "rankingEligible": true,
      "overallRank": 216,
      "url": "https://benchlm.ai/models/phi-4",
      "markdownUrl": "https://benchlm.ai/md/models/phi-4.md",
      "id": 128,
      "releaseDate": "2025-01-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "phi-4",
        "familyName": "Phi-4",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "phi-4",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 216,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 0
        },
        "coding": {
          "aaSciCode": 26
        },
        "reasoning": {
          "lcr": 0,
          "critpt": 0
        },
        "multimodalGrounded": {},
        "knowledge": {
          "artificialAnalysis": 4.55,
          "aaGpqaDiamond": 57.5,
          "aaHle": 3.8,
          "aaOmniscienceIndex": -55.7,
          "omniscienceAccuracy": 14.1,
          "omniscienceHallucinationRate": 81.2
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 23.5
        },
        "math": {}
      }
    },
    {
      "slug": "deepseekmath-v2",
      "canonicalModelKey": "deepseekmath-v2",
      "model": "DeepSeekMath V2",
      "creator": "DeepSeek",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 50.54,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 39.02,
        "upper": 62.05
      },
      "rankingEligible": true,
      "overallRank": 133,
      "url": "https://benchlm.ai/models/deepseekmath-v2",
      "markdownUrl": "https://benchlm.ai/md/models/deepseekmath-v2.md",
      "id": 62,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "deepseekmath",
        "familyName": "DeepSeekMath",
        "variantType": "snapshot",
        "snapshotLabel": "V2",
        "baseFamilyModelKey": "deepseekmath-v2",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 133,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "seed-2-0-lite",
      "canonicalModelKey": "seed-2-0-lite",
      "model": "Seed-2.0-Lite",
      "creator": "ByteDance",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": 50.49,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 38.97,
        "upper": 62
      },
      "rankingEligible": true,
      "overallRank": 136,
      "url": "https://benchlm.ai/models/seed-2-0-lite",
      "markdownUrl": "https://benchlm.ai/md/models/seed-2-0-lite.md",
      "id": 185,
      "releaseDate": "2026-03-10",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "seed-2-0",
        "familyName": "Seed 2.0",
        "variantType": "lite",
        "snapshotLabel": null,
        "baseFamilyModelKey": "seed-2-0-lite",
        "relatedModelKeys": [
          "seed-2-0-mini"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 136,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 2,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "ministral-3-14b-reasoning",
      "canonicalModelKey": "ministral-3-14b-reasoning",
      "model": "Ministral 3 14B (Reasoning)",
      "creator": "Mistral",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 49.99,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 38.47,
        "upper": 61.5
      },
      "rankingEligible": true,
      "overallRank": 138,
      "url": "https://benchlm.ai/models/ministral-3-14b-reasoning",
      "markdownUrl": "https://benchlm.ai/md/models/ministral-3-14b-reasoning.md",
      "id": 188,
      "releaseDate": "2025-12-02",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "ministral-3-14b",
        "familyName": "Ministral 3 14B",
        "variantType": "reasoning",
        "snapshotLabel": null,
        "baseFamilyModelKey": "ministral-3-14b",
        "relatedModelKeys": [
          "ministral-3-14b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 138,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 24,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "gpt-5-mini",
      "canonicalModelKey": "gpt-5-mini",
      "model": "GPT-5 mini",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 42.97,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 39.53,
        "upper": 46.41
      },
      "rankingEligible": true,
      "overallRank": 180,
      "url": "https://benchlm.ai/models/gpt-5-mini",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-5-mini.md",
      "id": 182,
      "releaseDate": "2025-08-07",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-5",
        "familyName": "GPT-5",
        "variantType": "mini",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-5-high",
        "relatedModelKeys": [
          "gpt-5-high",
          "gpt-5-medium",
          "gpt-5-nano"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 180,
        "categoryRanks": {
          "agentic": 105,
          "coding": 128
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 50,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {
          "vibeCodeBench": 14.171
        },
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {},
        "korean": {}
      }
    },
    {
      "slug": "o3-pro",
      "canonicalModelKey": "o3-pro",
      "model": "o3-pro",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": 47.2,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 42.84,
        "upper": 51.55
      },
      "rankingEligible": true,
      "overallRank": 155,
      "url": "https://benchlm.ai/models/o3-pro",
      "markdownUrl": "https://benchlm.ai/md/models/o3-pro.md",
      "id": 39,
      "releaseDate": "2025-04-16",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "o3",
        "familyName": "o3",
        "variantType": "pro",
        "snapshotLabel": null,
        "baseFamilyModelKey": "o3",
        "relatedModelKeys": [
          "o3",
          "o3-mini"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 155,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {
          "artificialAnalysis": 33.29,
          "aaGpqaDiamond": 84.5
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "ministral-3-14b",
      "canonicalModelKey": "ministral-3-14b",
      "model": "Ministral 3 14B",
      "creator": "Mistral",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 34,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 23,
        "upper": 45.01
      },
      "rankingEligible": true,
      "overallRank": 209,
      "url": "https://benchlm.ai/models/ministral-3-14b",
      "markdownUrl": "https://benchlm.ai/md/models/ministral-3-14b.md",
      "id": 192,
      "releaseDate": "2025-12-02",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "ministral-3-14b",
        "familyName": "Ministral 3 14B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "ministral-3-14b",
        "relatedModelKeys": [
          "ministral-3-14b-reasoning"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 209,
        "categoryRanks": {
          "agentic": 133,
          "coding": 137
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 18,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "aion-2-0",
      "canonicalModelKey": "aion-2-0",
      "model": "Aion-2.0",
      "creator": "Aion Labs",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 49.29,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 37.78,
        "upper": 60.81
      },
      "rankingEligible": true,
      "overallRank": 140,
      "url": "https://benchlm.ai/models/aion-2-0",
      "markdownUrl": "https://benchlm.ai/md/models/aion-2-0.md",
      "id": 190,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "aion-2-0",
        "familyName": "Aion-2.0",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "aion-2-0",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 140,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 2,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "mixtral-8x22b-instruct-v0-1",
      "canonicalModelKey": "mixtral-8x22b-instruct-v0-1",
      "model": "Mixtral 8x22B Instruct v0.1",
      "creator": "Mistral",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "64K",
      "contextWindowTokens": 64000,
      "displayScore": 49.19,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 37.68,
        "upper": 60.7
      },
      "rankingEligible": true,
      "overallRank": 142,
      "url": "https://benchlm.ai/models/mixtral-8x22b-instruct-v0-1",
      "markdownUrl": "https://benchlm.ai/md/models/mixtral-8x22b-instruct-v0-1.md",
      "id": 158,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "mixtral-8x22b",
        "familyName": "Mixtral 8x22B",
        "variantType": "instruct",
        "snapshotLabel": "v0.1",
        "baseFamilyModelKey": "mixtral-8x22b-instruct-v0-1",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 142,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "grok-code-fast-1",
      "canonicalModelKey": "grok-code-fast-1",
      "model": "Grok Code Fast 1",
      "creator": "xAI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": 37.73,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 34.7,
        "upper": 40.75
      },
      "rankingEligible": true,
      "overallRank": 203,
      "url": "https://benchlm.ai/models/grok-code-fast-1",
      "markdownUrl": "https://benchlm.ai/md/models/grok-code-fast-1.md",
      "id": 93,
      "releaseDate": "2025-08-28",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "grok-code-fast-1",
        "familyName": "Grok Code Fast 1",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "grok-code-fast-1",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 203,
        "categoryRanks": {
          "coding": 138
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 75.7
        },
        "coding": {
          "sweVerified": 70.8,
          "aaSciCode": 36.2
        },
        "reasoning": {
          "lcr": 51,
          "critpt": 0
        },
        "multimodalGrounded": {},
        "knowledge": {
          "artificialAnalysis": 21.95,
          "aaGpqaDiamond": 72.7,
          "aaHle": 8,
          "aaOmniscienceIndex": -37.1,
          "omniscienceAccuracy": 23.5,
          "omniscienceHallucinationRate": 79.3
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 41.4
        },
        "math": {}
      }
    },
    {
      "slug": "gpt-4-turbo",
      "canonicalModelKey": "gpt-4-turbo",
      "model": "GPT-4 Turbo",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 26.85,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 19.33,
        "upper": 34.36
      },
      "rankingEligible": true,
      "overallRank": 211,
      "url": "https://benchlm.ai/models/gpt-4-turbo",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-4-turbo.md",
      "id": 135,
      "releaseDate": "2023-11-06",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-4-turbo",
        "familyName": "GPT-4 Turbo",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-4-turbo",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 211,
        "categoryRanks": {
          "coding": 141
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {
          "aaCodingIndex": 21.49,
          "aaSciCode": 31.9
        },
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {
          "artificialAnalysis": 7.69,
          "aaHle": 3.1
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {},
        "korean": {}
      }
    },
    {
      "slug": "z-1",
      "canonicalModelKey": "z-1",
      "model": "Z-1",
      "creator": "Z",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 45.73,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 34.21,
        "upper": 57.24
      },
      "rankingEligible": true,
      "overallRank": 163,
      "url": "https://benchlm.ai/models/z-1",
      "markdownUrl": "https://benchlm.ai/md/models/z-1.md",
      "id": 138,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "z-1",
        "familyName": "Z-1",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "z-1",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 163,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "seed-1-6-flash",
      "canonicalModelKey": "seed-1-6-flash",
      "model": "Seed 1.6 Flash",
      "creator": "ByteDance",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": 45.68,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 34.17,
        "upper": 57.19
      },
      "rankingEligible": true,
      "overallRank": 166,
      "url": "https://benchlm.ai/models/seed-1-6-flash",
      "markdownUrl": "https://benchlm.ai/md/models/seed-1-6-flash.md",
      "id": 191,
      "releaseDate": "2025-06-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "seed-1-6",
        "familyName": "Seed 1.6",
        "variantType": "flash",
        "snapshotLabel": null,
        "baseFamilyModelKey": "seed-1-6",
        "relatedModelKeys": [
          "seed-1-6"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 166,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 2,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "nemotron-4-15b",
      "canonicalModelKey": "nemotron-4-15b",
      "model": "Nemotron-4 15B",
      "creator": "NVIDIA",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": 45.68,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 34.17,
        "upper": 57.19
      },
      "rankingEligible": true,
      "overallRank": 165,
      "url": "https://benchlm.ai/models/nemotron-4-15b",
      "markdownUrl": "https://benchlm.ai/md/models/nemotron-4-15b.md",
      "id": 140,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "nemotron-4-15b",
        "familyName": "Nemotron-4 15B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "nemotron-4-15b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 165,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "mistral-8x7b",
      "canonicalModelKey": "mistral-8x7b",
      "model": "Mistral 8x7B",
      "creator": "Mistral",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": 45.58,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 34.07,
        "upper": 57.1
      },
      "rankingEligible": true,
      "overallRank": 167,
      "url": "https://benchlm.ai/models/mistral-8x7b",
      "markdownUrl": "https://benchlm.ai/md/models/mistral-8x7b.md",
      "id": 133,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "mistral-8x7b",
        "familyName": "Mistral 8x7B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "mistral-8x7b",
        "relatedModelKeys": [
          "mistral-8x7b-v0-2"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 167,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "moonshot-v1",
      "canonicalModelKey": "moonshot-v1",
      "model": "Moonshot v1",
      "creator": "Moonshot AI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 45.38,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 33.87,
        "upper": 56.9
      },
      "rankingEligible": true,
      "overallRank": 168,
      "url": "https://benchlm.ai/models/moonshot-v1",
      "markdownUrl": "https://benchlm.ai/md/models/moonshot-v1.md",
      "id": 137,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "moonshot",
        "familyName": "Moonshot",
        "variantType": "snapshot",
        "snapshotLabel": "v1",
        "baseFamilyModelKey": "moonshot-v1",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 168,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "seed-2-0-mini",
      "canonicalModelKey": "seed-2-0-mini",
      "model": "Seed-2.0-Mini",
      "creator": "ByteDance",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": 45.19,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 33.67,
        "upper": 56.7
      },
      "rankingEligible": true,
      "overallRank": 169,
      "url": "https://benchlm.ai/models/seed-2-0-mini",
      "markdownUrl": "https://benchlm.ai/md/models/seed-2-0-mini.md",
      "id": 193,
      "releaseDate": "2026-03-10",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "seed-2-0",
        "familyName": "Seed 2.0",
        "variantType": "mini",
        "snapshotLabel": null,
        "baseFamilyModelKey": "seed-2-0-lite",
        "relatedModelKeys": [
          "seed-2-0-lite"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 169,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 2,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "nemotron-ultra-253b",
      "canonicalModelKey": "nemotron-ultra-253b",
      "model": "Nemotron Ultra 253B",
      "creator": "NVIDIA",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": 44.99,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 33.48,
        "upper": 56.5
      },
      "rankingEligible": true,
      "overallRank": 171,
      "url": "https://benchlm.ai/models/nemotron-ultra-253b",
      "markdownUrl": "https://benchlm.ai/md/models/nemotron-ultra-253b.md",
      "id": 132,
      "releaseDate": "2026-02-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "nemotron-ultra-253b",
        "familyName": "Nemotron Ultra 253B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "nemotron-ultra-253b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 171,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 11.4
        },
        "coding": {
          "aaSciCode": 34.7
        },
        "reasoning": {
          "lcr": 7.7,
          "critpt": 0
        },
        "multimodalGrounded": {},
        "knowledge": {
          "artificialAnalysis": 8.92,
          "aaGpqaDiamond": 72.8,
          "aaHle": 7.4,
          "aaOmniscienceIndex": -44.9,
          "omniscienceAccuracy": 20.1,
          "omniscienceHallucinationRate": 81.2
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 38.2
        },
        "math": {}
      }
    },
    {
      "slug": "gemini-1-0-pro",
      "canonicalModelKey": "gemini-1-0-pro",
      "model": "Gemini 1.0 Pro",
      "creator": "Google",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": 21.28,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 13.82,
        "upper": 28.75
      },
      "rankingEligible": true,
      "overallRank": 217,
      "url": "https://benchlm.ai/models/gemini-1-0-pro",
      "markdownUrl": "https://benchlm.ai/md/models/gemini-1-0-pro.md",
      "id": 146,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gemini-1-0-pro",
        "familyName": "Gemini 1.0 Pro",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gemini-1-0-pro",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 217,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {
          "aaSciCode": 11.7
        },
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {
          "artificialAnalysis": 2.74,
          "aaGpqaDiamond": 27.7,
          "aaHle": 4.2
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "claude-3-haiku",
      "canonicalModelKey": "claude-3-haiku",
      "model": "Claude 3 Haiku",
      "creator": "Anthropic",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": 20.86,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 14.72,
        "upper": 27.01
      },
      "rankingEligible": true,
      "overallRank": 218,
      "url": "https://benchlm.ai/models/claude-3-haiku",
      "markdownUrl": "https://benchlm.ai/md/models/claude-3-haiku.md",
      "id": 130,
      "releaseDate": "2024-03-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "claude-3-haiku",
        "familyName": "Claude 3 Haiku",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "claude-3-haiku",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 218,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 21.1
        },
        "coding": {
          "aaSciCode": 18.6
        },
        "reasoning": {
          "lcr": 25,
          "critpt": 0
        },
        "multimodalGrounded": {
          "aaMmmuPro": 30.8
        },
        "knowledge": {
          "artificialAnalysis": 3.5,
          "aaGpqaDiamond": 37.4,
          "aaHle": 4.1,
          "aaOmniscienceIndex": -48.6,
          "omniscienceAccuracy": 17.6,
          "omniscienceHallucinationRate": 80.5
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 36.1
        },
        "math": {}
      }
    },
    {
      "slug": "lfm2-24b-a2b",
      "canonicalModelKey": "lfm2-24b-a2b",
      "model": "LFM2-24B-A2B",
      "creator": "LiquidAI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": 18.29,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 15.23,
        "upper": 21.35
      },
      "rankingEligible": true,
      "overallRank": 221,
      "url": "https://benchlm.ai/models/lfm2-24b-a2b",
      "markdownUrl": "https://benchlm.ai/md/models/lfm2-24b-a2b.md",
      "id": 196,
      "releaseDate": "2026-01-10",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "lfm2-24b-a2b",
        "familyName": "LFM2-24B-A2B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "lfm2-24b-a2b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 221,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 13,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "claude-4-1-opus-thinking",
      "canonicalModelKey": "claude-4-1-opus-thinking",
      "model": "Claude 4.1 Opus Thinking",
      "creator": "Anthropic",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": 35.27,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 16.26,
        "upper": 54.27
      },
      "rankingEligible": true,
      "overallRank": 206,
      "url": "https://benchlm.ai/models/claude-4-1-opus-thinking",
      "markdownUrl": "https://benchlm.ai/md/models/claude-4-1-opus-thinking.md",
      "id": 98,
      "releaseDate": "2025-08-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "claude-4-1-opus",
        "familyName": "Claude 4.1 Opus",
        "variantType": "reasoning",
        "snapshotLabel": null,
        "baseFamilyModelKey": "claude-4-1-opus",
        "relatedModelKeys": [
          "claude-4-1-opus"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 206,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 71.4
        },
        "coding": {
          "aaSciCode": 40.9
        },
        "reasoning": {
          "lcr": 73.3,
          "critpt": 0
        },
        "multimodalGrounded": {
          "aaMmmuPro": 67.9
        },
        "knowledge": {
          "artificialAnalysis": 34.54,
          "aaGpqaDiamond": 80.9,
          "aaHle": 12.5
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 55.4
        },
        "math": {}
      }
    },
    {
      "slug": "ministral-3-8b-reasoning",
      "canonicalModelKey": "ministral-3-8b-reasoning",
      "model": "Ministral 3 8B (Reasoning)",
      "creator": "Mistral",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 40.94,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 29.42,
        "upper": 52.45
      },
      "rankingEligible": true,
      "overallRank": 192,
      "url": "https://benchlm.ai/models/ministral-3-8b-reasoning",
      "markdownUrl": "https://benchlm.ai/md/models/ministral-3-8b-reasoning.md",
      "id": 197,
      "releaseDate": "2025-12-02",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "ministral-3-8b",
        "familyName": "Ministral 3 8B",
        "variantType": "reasoning",
        "snapshotLabel": null,
        "baseFamilyModelKey": "ministral-3-8b",
        "relatedModelKeys": [
          "ministral-3-8b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 192,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 18,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "nova-pro",
      "canonicalModelKey": "nova-pro",
      "model": "Nova Pro",
      "creator": "Amazon",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 19.67,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 16.55,
        "upper": 22.79
      },
      "rankingEligible": true,
      "overallRank": 220,
      "url": "https://benchlm.ai/models/nova-pro",
      "markdownUrl": "https://benchlm.ai/md/models/nova-pro.md",
      "id": 162,
      "releaseDate": "2025-04-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "nova-pro",
        "familyName": "Nova Pro",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "nova-pro",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 220,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 14
        },
        "coding": {
          "aaSciCode": 20.8
        },
        "reasoning": {
          "lcr": 22.3,
          "critpt": 0
        },
        "multimodalGrounded": {
          "aaMmmuPro": 44.3
        },
        "knowledge": {
          "artificialAnalysis": 7.46,
          "aaGpqaDiamond": 49.9,
          "aaHle": 3.2,
          "aaOmniscienceIndex": -47.7,
          "omniscienceAccuracy": 16.9,
          "omniscienceHallucinationRate": 77.7
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 38.1
        },
        "math": {}
      }
    },
    {
      "slug": "ministral-3-8b",
      "canonicalModelKey": "ministral-3-8b",
      "model": "Ministral 3 8B",
      "creator": "Mistral",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 20.38,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 16.69,
        "upper": 24.07
      },
      "rankingEligible": true,
      "overallRank": 219,
      "url": "https://benchlm.ai/models/ministral-3-8b",
      "markdownUrl": "https://benchlm.ai/md/models/ministral-3-8b.md",
      "id": 199,
      "releaseDate": "2025-12-02",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "ministral-3-8b",
        "familyName": "Ministral 3 8B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "ministral-3-8b",
        "relatedModelKeys": [
          "ministral-3-8b-reasoning"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 219,
        "categoryRanks": {
          "agentic": 139,
          "coding": 145
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 18,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "mistral-7b-v0-3",
      "canonicalModelKey": "mistral-7b-v0-3",
      "model": "Mistral 7B v0.3",
      "creator": "Mistral",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": 40.45,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 28.94,
        "upper": 51.97
      },
      "rankingEligible": true,
      "overallRank": 194,
      "url": "https://benchlm.ai/models/mistral-7b-v0-3",
      "markdownUrl": "https://benchlm.ai/md/models/mistral-7b-v0-3.md",
      "id": 169,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "mistral-7b",
        "familyName": "Mistral 7B",
        "variantType": "snapshot",
        "snapshotLabel": "v0.3",
        "baseFamilyModelKey": "mistral-7b-v0-3",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 194,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "qwen2-5-vl-32b",
      "canonicalModelKey": "qwen2-5-vl-32b",
      "model": "Qwen2.5-VL-32B",
      "creator": "Alibaba",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": 40.4,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 28.89,
        "upper": 51.92
      },
      "rankingEligible": true,
      "overallRank": 195,
      "url": "https://benchlm.ai/models/qwen2-5-vl-32b",
      "markdownUrl": "https://benchlm.ai/md/models/qwen2-5-vl-32b.md",
      "id": 107,
      "releaseDate": "2025-01-26",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "qwen2-5-vl-32b",
        "familyName": "Qwen2.5-VL-32B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "qwen2-5-vl-32b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 195,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "llama-4-behemoth",
      "canonicalModelKey": "llama-4-behemoth",
      "model": "Llama 4 Behemoth",
      "creator": "Meta",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": 40.36,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 28.84,
        "upper": 51.87
      },
      "rankingEligible": true,
      "overallRank": 196,
      "url": "https://benchlm.ai/models/llama-4-behemoth",
      "markdownUrl": "https://benchlm.ai/md/models/llama-4-behemoth.md",
      "id": 156,
      "releaseDate": "2026-02-28",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "llama-4-behemoth",
        "familyName": "Llama 4 Behemoth",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "llama-4-behemoth",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 196,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "lfm2-5-1-2b-thinking",
      "canonicalModelKey": "lfm2-5-1-2b-thinking",
      "model": "LFM2.5-1.2B-Thinking",
      "creator": "LiquidAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": 15.71,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 12.66,
        "upper": 18.75
      },
      "rankingEligible": true,
      "overallRank": 223,
      "url": "https://benchlm.ai/models/lfm2-5-1-2b-thinking",
      "markdownUrl": "https://benchlm.ai/md/models/lfm2-5-1-2b-thinking.md",
      "id": 198,
      "releaseDate": "2026-03-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "lfm2-5-1-2b",
        "familyName": "LFM2.5 1.2B",
        "variantType": "reasoning",
        "snapshotLabel": null,
        "baseFamilyModelKey": "lfm2-5-1-2b-instruct",
        "relatedModelKeys": [
          "lfm2-5-1-2b-instruct"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 223,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 13,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "ministral-3-3b-reasoning",
      "canonicalModelKey": "ministral-3-3b-reasoning",
      "model": "Ministral 3 3B (Reasoning)",
      "creator": "Mistral",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 40.07,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 28.55,
        "upper": 51.58
      },
      "rankingEligible": true,
      "overallRank": 198,
      "url": "https://benchlm.ai/models/ministral-3-3b-reasoning",
      "markdownUrl": "https://benchlm.ai/md/models/ministral-3-3b-reasoning.md",
      "id": 200,
      "releaseDate": "2025-12-02",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "ministral-3-3b",
        "familyName": "Ministral 3 3B",
        "variantType": "reasoning",
        "snapshotLabel": null,
        "baseFamilyModelKey": "ministral-3-3b",
        "relatedModelKeys": [
          "ministral-3-3b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 198,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 18,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "minimax-m1-80k",
      "canonicalModelKey": "minimax-m1-80k",
      "model": "MiniMax M1 80k",
      "creator": "MiniMax",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "80K",
      "contextWindowTokens": 80000,
      "displayScore": 24.21,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 14.31,
        "upper": 34.12
      },
      "rankingEligible": true,
      "overallRank": 214,
      "url": "https://benchlm.ai/models/minimax-m1-80k",
      "markdownUrl": "https://benchlm.ai/md/models/minimax-m1-80k.md",
      "id": 195,
      "releaseDate": "2025-01-15",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "minimax-m1-80k",
        "familyName": "MiniMax M1 80k",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "minimax-m1-80k",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 214,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 35,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "ministral-3-3b",
      "canonicalModelKey": "ministral-3-3b",
      "model": "Ministral 3 3B",
      "creator": "Mistral",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 17.93,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 14,
        "upper": 21.86
      },
      "rankingEligible": true,
      "overallRank": 222,
      "url": "https://benchlm.ai/models/ministral-3-3b",
      "markdownUrl": "https://benchlm.ai/md/models/ministral-3-3b.md",
      "id": 202,
      "releaseDate": "2025-12-02",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "ministral-3-3b",
        "familyName": "Ministral 3 3B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "ministral-3-3b",
        "relatedModelKeys": [
          "ministral-3-3b-reasoning"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 222,
        "categoryRanks": {
          "agentic": 140,
          "coding": 146
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 18,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "mistral-8x7b-v0-2",
      "canonicalModelKey": "mistral-8x7b-v0-2",
      "model": "Mistral 8x7B v0.2",
      "creator": "Mistral",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": 39.68,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 28.17,
        "upper": 51.19
      },
      "rankingEligible": true,
      "overallRank": 199,
      "url": "https://benchlm.ai/models/mistral-8x7b-v0-2",
      "markdownUrl": "https://benchlm.ai/md/models/mistral-8x7b-v0-2.md",
      "id": 170,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "mistral-8x7b",
        "familyName": "Mistral 8x7B",
        "variantType": "snapshot",
        "snapshotLabel": "v0.2",
        "baseFamilyModelKey": "mistral-8x7b",
        "relatedModelKeys": [
          "mistral-8x7b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 199,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "lfm2-5-1-2b-instruct",
      "canonicalModelKey": "lfm2-5-1-2b-instruct",
      "model": "LFM2.5-1.2B-Instruct",
      "creator": "LiquidAI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": 14.99,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "supported",
      "scoreInterval90": {
        "lower": 12.08,
        "upper": 17.9
      },
      "rankingEligible": true,
      "overallRank": 224,
      "url": "https://benchlm.ai/models/lfm2-5-1-2b-instruct",
      "markdownUrl": "https://benchlm.ai/md/models/lfm2-5-1-2b-instruct.md",
      "id": 201,
      "releaseDate": "2026-03-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "lfm2-5-1-2b",
        "familyName": "LFM2.5 1.2B",
        "variantType": "instruct",
        "snapshotLabel": null,
        "baseFamilyModelKey": "lfm2-5-1-2b-instruct",
        "relatedModelKeys": [
          "lfm2-5-1-2b-thinking"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 224,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 13,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "ling-3-0-flash-fin",
      "canonicalModelKey": "ling-3-0-flash-fin",
      "model": "Ling 3.0 Flash Fin",
      "creator": "InclusionAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/ling-3-0-flash-fin",
      "markdownUrl": "https://benchlm.ai/md/models/ling-3-0-flash-fin.md",
      "id": 411,
      "releaseDate": "2026-08-28",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "ling-3-0",
        "familyName": "Ling 3.0",
        "variantType": "flash-fin",
        "snapshotLabel": "Fin",
        "baseFamilyModelKey": "ling-3-0-flash",
        "relatedModelKeys": [
          "ling-3-0-flash",
          "ling-3-0-flash-fp8"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 3,
        "verifiedBenchmarkCount": 3,
        "rankableBenchmarkCount": 3,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "financeAgentV2": 59.81,
          "apexAgents": 29.17,
          "spreadsheetBench2": 21.81
        },
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "kimi-k2-7-code",
      "canonicalModelKey": "kimi-k2-7-code",
      "model": "Kimi K2.7 Code",
      "creator": "Moonshot AI",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": 54.01,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 42.99,
        "upper": 65.03
      },
      "rankingEligible": true,
      "overallRank": 110,
      "url": "https://benchlm.ai/models/kimi-k2-7-code",
      "markdownUrl": "https://benchlm.ai/md/models/kimi-k2-7-code.md",
      "id": 258,
      "releaseDate": "2026-06-12",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "kimi-k2-7-code",
        "familyName": "Kimi K2.7 Code",
        "variantType": "code",
        "snapshotLabel": null,
        "baseFamilyModelKey": "kimi-k2-7-code",
        "relatedModelKeys": [
          "kimi-2-6",
          "kimi-k2-5"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "kimi-2-6"
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": 61.6,
          "coding": 59.8,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 110,
        "categoryRanks": {
          "agentic": 111,
          "coding": 55
        },
        "categoryRankingEligible": {
          "agentic": true,
          "coding": true,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 3,
        "verifiedBenchmarkCount": 3,
        "rankableBenchmarkCount": 3,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "kimiClaw247": 46.9,
          "mcpAtlas": 76,
          "mcpMarkVerified": 81.1,
          "aaAgenticIndex": 30.28,
          "tau2Bench": 90.1,
          "gdpvalAaNormalized": 34.3,
          "gdpvalAa": 1187
        },
        "coding": {
          "kimiCodeBenchV2": 62,
          "programBench": 53.6,
          "mlsBenchLite": 35.1,
          "cursorBench32": 49.7,
          "aaCodingIndex": 60.76,
          "aaSciCode": 47.5,
          "openHarmonyBench": 52.1
        },
        "reasoning": {
          "lcr": 75,
          "critpt": 10
        },
        "multimodalGrounded": {
          "designArenaWebsite": 1287
        },
        "knowledge": {
          "artificialAnalysis": 43.02,
          "aaGpqaDiamond": 89.6,
          "aaHle": 35,
          "aaOmniscienceIndex": -10.2,
          "omniscienceAccuracy": 39.6,
          "omniscienceHallucinationRate": 82.4
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 63.1
        },
        "math": {}
      }
    },
    {
      "slug": "lfm2-5-vl-1-6b-extract",
      "canonicalModelKey": "lfm2-5-vl-1-6b-extract",
      "model": "LFM2.5-VL-1.6B-Extract",
      "creator": "LiquidAI",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/lfm2-5-vl-1-6b-extract",
      "markdownUrl": "https://benchlm.ai/md/models/lfm2-5-vl-1-6b-extract.md",
      "id": 256,
      "releaseDate": "2026-05-26",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "lfm2-5-vl-extract",
        "familyName": "LFM2.5-VL Extract",
        "variantType": "1-6b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "lfm2-5-vl-1-6b-extract",
        "relatedModelKeys": [
          "lfm2-5-vl-450m-extract",
          "lfm2-5-vl-450m"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 3,
        "verifiedBenchmarkCount": 3,
        "rankableBenchmarkCount": 3,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 8.5
        },
        "coding": {
          "aaSciCode": 3
        },
        "reasoning": {
          "lcr": 0,
          "critpt": 0
        },
        "multimodalGrounded": {
          "liquidExtractJsonValidity": 99.6,
          "liquidExtractSchemaF1": 99.6,
          "liquidExtractVlmJudge": 90.6,
          "aaMmmuPro": 26.5
        },
        "knowledge": {
          "artificialAnalysis": 1,
          "aaGpqaDiamond": 28.9,
          "aaHle": 5.1,
          "aaOmniscienceIndex": -84.4,
          "omniscienceAccuracy": 5.8,
          "omniscienceHallucinationRate": 95.7
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 33.1
        },
        "math": {}
      }
    },
    {
      "slug": "lfm2-5-vl-450m-extract",
      "canonicalModelKey": "lfm2-5-vl-450m-extract",
      "model": "LFM2.5-VL-450M-Extract",
      "creator": "LiquidAI",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/lfm2-5-vl-450m-extract",
      "markdownUrl": "https://benchlm.ai/md/models/lfm2-5-vl-450m-extract.md",
      "id": 255,
      "releaseDate": "2026-05-26",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "lfm2-5-vl-extract",
        "familyName": "LFM2.5-VL Extract",
        "variantType": "450m",
        "snapshotLabel": null,
        "baseFamilyModelKey": "lfm2-5-vl-1-6b-extract",
        "relatedModelKeys": [
          "lfm2-5-vl-1-6b-extract",
          "lfm2-5-vl-450m"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 3,
        "verifiedBenchmarkCount": 3,
        "rankableBenchmarkCount": 3,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {
          "liquidExtractJsonValidity": 98.9,
          "liquidExtractSchemaF1": 98.8,
          "liquidExtractVlmJudge": 84.5
        },
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "apodex-1-1-mini",
      "canonicalModelKey": "apodex-1-1-mini",
      "model": "Apodex 1.1 Mini",
      "creator": "Apodex",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/apodex-1-1-mini",
      "markdownUrl": "https://benchlm.ai/md/models/apodex-1-1-mini.md",
      "id": 407,
      "releaseDate": "2026-08-24",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "apodex-1-1",
        "familyName": "Apodex 1.1",
        "variantType": "mini-35b-a3b",
        "snapshotLabel": "35B-A3B",
        "baseFamilyModelKey": "apodex-1-1",
        "relatedModelKeys": [
          "apodex-1-1"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 2,
        "verifiedBenchmarkCount": 2,
        "rankableBenchmarkCount": 2,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "apexAgents": 27.7,
          "aaAgenticIndex": 36.64,
          "apexAgentsAa": 2.9,
          "gdpvalAaNormalized": 42.3,
          "gdpvalAa": 1347
        },
        "coding": {
          "aaCodingIndex": 60.76,
          "aaSciCode": 42.9
        },
        "reasoning": {
          "lcr": 74.7,
          "critpt": 4.6
        },
        "multimodalGrounded": {
          "aaMmmuPro": 79.2
        },
        "knowledge": {
          "frontierScienceResearch": 51.7,
          "artificialAnalysis": 44,
          "aaGpqaDiamond": 86.4,
          "aaHle": 34.1,
          "aaOmniscienceIndex": -21.9,
          "omniscienceAccuracy": 31.7,
          "omniscienceHallucinationRate": 78.4
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "gpt-5-1-codex",
      "canonicalModelKey": "gpt-5-1-codex",
      "model": "GPT-5.1-Codex",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "400K",
      "contextWindowTokens": 400000,
      "displayScore": 52.63,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 41.12,
        "upper": 64.14
      },
      "rankingEligible": true,
      "overallRank": 119,
      "url": "https://benchlm.ai/models/gpt-5-1-codex",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-5-1-codex.md",
      "id": 231,
      "releaseDate": "2025-10-15",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-5-1-codex",
        "familyName": "GPT-5.1-Codex",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-5-1-codex",
        "relatedModelKeys": [
          "gpt-5-1-codex-max"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 119,
        "categoryRanks": {
          "agentic": 56,
          "coding": 88
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 2,
        "verifiedBenchmarkCount": 2,
        "rankableBenchmarkCount": 2,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 83,
          "gertLabs": 49.68,
          "jobBench": 26.2
        },
        "coding": {
          "vibeCodeBench": 13.115,
          "aaSciCode": 40.2
        },
        "reasoning": {
          "lcr": 69,
          "critpt": 5.7
        },
        "multimodalGrounded": {
          "aaMmmuPro": 72.5,
          "designArenaWebsite": 1180
        },
        "knowledge": {
          "artificialAnalysis": 35.6,
          "aaGpqaDiamond": 86,
          "aaHle": 25.7,
          "aaOmniscienceIndex": -6.5,
          "omniscienceAccuracy": 39.9,
          "omniscienceHallucinationRate": 77.2
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 70
        },
        "math": {}
      }
    },
    {
      "slug": "claude-mythos-5-1",
      "canonicalModelKey": "claude-mythos-5-1",
      "model": "Claude Mythos 5.1",
      "creator": "Anthropic",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/claude-mythos-5-1",
      "markdownUrl": "https://benchlm.ai/md/models/claude-mythos-5-1.md",
      "id": 417,
      "releaseDate": "2026-09-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "claude-mythos",
        "familyName": "Claude Mythos",
        "variantType": "restricted",
        "snapshotLabel": null,
        "baseFamilyModelKey": "claude-mythos-5-1",
        "relatedModelKeys": [
          "claude-fable-5-1",
          "claude-mythos-5"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "claude-mythos-5"
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "terminalBench4": 60.9
        },
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "sakana-fugu-cyber",
      "canonicalModelKey": "sakana-fugu-cyber",
      "model": "Fugu Cyber",
      "creator": "Sakana AI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/sakana-fugu-cyber",
      "markdownUrl": "https://benchlm.ai/md/models/sakana-fugu-cyber.md",
      "id": 290,
      "releaseDate": "2026-07-21",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "sakana-fugu",
        "familyName": "Sakana Fugu",
        "variantType": "cyber",
        "snapshotLabel": "Cyber",
        "baseFamilyModelKey": "sakana-fugu-ultra",
        "relatedModelKeys": [
          "sakana-fugu-ultra",
          "sakana-fugu"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "cyberGym": 86.9,
          "ctiRealm": 72.1
        },
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "qwen3-8-max-preview",
      "canonicalModelKey": "qwen3-8-max-preview",
      "model": "Qwen3.8 Max Preview",
      "creator": "Alibaba",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": null,
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/qwen3-8-max-preview",
      "markdownUrl": "https://benchlm.ai/md/models/qwen3-8-max-preview.md",
      "id": 285,
      "releaseDate": "2026-07-19",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "qwen3-8-max",
        "familyName": "Qwen3.8 Max",
        "variantType": "preview",
        "snapshotLabel": "preview",
        "baseFamilyModelKey": "qwen3-8-max",
        "relatedModelKeys": [
          "qwen3-8-max",
          "qwen3-7-max",
          "qwen3-6-max-preview"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": "qwen3-7-max"
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "aaAgenticIndex": 58.4,
          "gdpvalAaNormalized": 61,
          "gdpvalAa": 1721
        },
        "coding": {
          "aaCodingIndex": 71.81,
          "aaSciCode": 52.9,
          "openHarmonyBench": 56
        },
        "reasoning": {
          "lcr": 74.3,
          "critpt": 20
        },
        "multimodalGrounded": {
          "aaMmmuPro": 82.3,
          "designArenaWebsite": 1299
        },
        "knowledge": {
          "artificialAnalysis": 58.08,
          "aaGpqaDiamond": 92.7,
          "aaHle": 43,
          "aaOmniscienceIndex": 3.4,
          "omniscienceAccuracy": 31.9,
          "omniscienceHallucinationRate": 41.7
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "holo3-1-35b-a3b",
      "canonicalModelKey": "holo3-1-35b-a3b",
      "model": "Holo3.1-35B-A3B",
      "creator": "H Company",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/holo3-1-35b-a3b",
      "markdownUrl": "https://benchlm.ai/md/models/holo3-1-35b-a3b.md",
      "id": 242,
      "releaseDate": "2026-06-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "holo3-1",
        "familyName": "Holo3.1",
        "variantType": "35b-a3b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "holo3-1-35b-a3b",
        "relatedModelKeys": [
          "holo3-1-35b-a3b-fp8",
          "holo3-1-35b-a3b-nvfp4",
          "holo3-1-35b-a3b-gguf",
          "holo3-1-9b",
          "holo3-1-4b",
          "holo3-1-0-8b"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "holo3-35b-a3b"
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "androidWorld": 79.3
        },
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "holo3-1-4b",
      "canonicalModelKey": "holo3-1-4b",
      "model": "Holo3.1-4B",
      "creator": "H Company",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/holo3-1-4b",
      "markdownUrl": "https://benchlm.ai/md/models/holo3-1-4b.md",
      "id": 247,
      "releaseDate": "2026-06-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "holo3-1",
        "familyName": "Holo3.1",
        "variantType": "4b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "holo3-1-35b-a3b",
        "relatedModelKeys": [
          "holo3-1-35b-a3b",
          "holo3-1-35b-a3b-fp8",
          "holo3-1-35b-a3b-nvfp4",
          "holo3-1-35b-a3b-gguf",
          "holo3-1-9b",
          "holo3-1-0-8b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": "holo2-4b"
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "androidWorld": 71
        },
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "holo3-1-9b",
      "canonicalModelKey": "holo3-1-9b",
      "model": "Holo3.1-9B",
      "creator": "H Company",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/holo3-1-9b",
      "markdownUrl": "https://benchlm.ai/md/models/holo3-1-9b.md",
      "id": 246,
      "releaseDate": "2026-06-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "holo3-1",
        "familyName": "Holo3.1",
        "variantType": "9b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "holo3-1-35b-a3b",
        "relatedModelKeys": [
          "holo3-1-35b-a3b",
          "holo3-1-35b-a3b-fp8",
          "holo3-1-35b-a3b-nvfp4",
          "holo3-1-35b-a3b-gguf",
          "holo3-1-4b",
          "holo3-1-0-8b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": "holo2-8b"
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "androidWorld": 71
        },
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "qwen3-max",
      "canonicalModelKey": "qwen3-max",
      "model": "Qwen3 Max",
      "creator": "Alibaba",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": 48.25,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 36.73,
        "upper": 59.76
      },
      "rankingEligible": true,
      "overallRank": 145,
      "url": "https://benchlm.ai/models/qwen3-max",
      "markdownUrl": "https://benchlm.ai/md/models/qwen3-max.md",
      "id": 233,
      "releaseDate": "2026-04-20",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "qwen3-max",
        "familyName": "Qwen3 Max",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "qwen3-max",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 145,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 74.3,
          "gertLabs": 43.74
        },
        "coding": {
          "vibeCodeBench": 3.506,
          "aaSciCode": 38.3
        },
        "reasoning": {
          "lcr": 48.3,
          "critpt": 0
        },
        "multimodalGrounded": {
          "designArenaWebsite": 1137
        },
        "knowledge": {
          "artificialAnalysis": 24.45,
          "aaGpqaDiamond": 76.4,
          "aaHle": 11.9,
          "aaOmniscienceIndex": -43.5,
          "omniscienceAccuracy": 24.4,
          "omniscienceHallucinationRate": 89.9
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 44.1
        },
        "math": {}
      }
    },
    {
      "slug": "claude-mythos-preview",
      "canonicalModelKey": "claude-mythos-preview",
      "model": "Claude Mythos Preview",
      "creator": "Anthropic",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": null,
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/claude-mythos-preview",
      "markdownUrl": "https://benchlm.ai/md/models/claude-mythos-preview.md",
      "id": 391,
      "releaseDate": "2026-04-07",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "claude-mythos",
        "familyName": "Claude Mythos",
        "variantType": "preview",
        "snapshotLabel": "preview",
        "baseFamilyModelKey": "claude-mythos-5",
        "relatedModelKeys": [
          "claude-mythos-5"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "cyberGym": 83.1,
          "exploitGym": 17.5
        },
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {},
        "external": {
          "exploitBench": 69,
          "sconePostCutoffSuccess": 100,
          "sconePostCutoffRevenueUsdM": 35
        }
      }
    },
    {
      "slug": "composer-2-fast",
      "canonicalModelKey": "composer-2-fast",
      "model": "Composer 2 Fast",
      "creator": "Cursor",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/composer-2-fast",
      "markdownUrl": "https://benchlm.ai/md/models/composer-2-fast.md",
      "id": 176,
      "releaseDate": "2026-03-19",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "composer",
        "familyName": "Composer",
        "variantType": "fast",
        "snapshotLabel": null,
        "baseFamilyModelKey": "composer-2",
        "relatedModelKeys": [
          "composer-2"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {
          "reactNativeEvals": 94.9
        },
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "claude-opus-4-5-thinking",
      "canonicalModelKey": "claude-opus-4-5-thinking",
      "model": "Claude Opus 4.5 Thinking",
      "creator": "Anthropic",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": 57.25,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 45.74,
        "upper": 68.77
      },
      "rankingEligible": true,
      "overallRank": 92,
      "url": "https://benchlm.ai/models/claude-opus-4-5-thinking",
      "markdownUrl": "https://benchlm.ai/md/models/claude-opus-4-5-thinking.md",
      "id": 230,
      "releaseDate": "2025-11-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "claude-opus-4-5",
        "familyName": "Claude Opus 4.5",
        "variantType": "reasoning",
        "snapshotLabel": "thinking",
        "baseFamilyModelKey": "claude-opus-4-5",
        "relatedModelKeys": [
          "claude-opus-4-5"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": 63.7,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 92,
        "categoryRanks": {
          "knowledge": 36
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": true,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 89.5
        },
        "coding": {
          "vibeCodeBench": 20.63,
          "aaSciCode": 49.5
        },
        "reasoning": {
          "lcr": 76,
          "critpt": 4.6
        },
        "multimodalGrounded": {
          "aaMmmuPro": 74,
          "designArenaWebsite": 1265
        },
        "knowledge": {
          "artificialAnalysis": 41.87,
          "aaGpqaDiamond": 86.6,
          "aaHle": 30.1,
          "aaOmniscienceIndex": 14,
          "omniscienceAccuracy": 46.6,
          "omniscienceHallucinationRate": 61,
          "aaMmluPro": 89.5
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 58
        },
        "math": {}
      }
    },
    {
      "slug": "claude-haiku-4-5-thinking",
      "canonicalModelKey": "claude-haiku-4-5-thinking",
      "model": "Claude Haiku 4.5 Thinking",
      "creator": "Anthropic",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/claude-haiku-4-5-thinking",
      "markdownUrl": "https://benchlm.ai/md/models/claude-haiku-4-5-thinking.md",
      "id": 232,
      "releaseDate": "2025-10-15",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "claude-haiku-4-5",
        "familyName": "Claude Haiku 4.5",
        "variantType": "reasoning",
        "snapshotLabel": "thinking",
        "baseFamilyModelKey": "claude-haiku-4-5",
        "relatedModelKeys": [
          "claude-haiku-4-5"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {
          "vibeCodeBench": 11.393
        },
        "reasoning": {},
        "multimodalGrounded": {
          "designArenaWebsite": 1140
        },
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "claude-sonnet-4-5-thinking",
      "canonicalModelKey": "claude-sonnet-4-5-thinking",
      "model": "Claude Sonnet 4.5 Thinking",
      "creator": "Anthropic",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/claude-sonnet-4-5-thinking",
      "markdownUrl": "https://benchlm.ai/md/models/claude-sonnet-4-5-thinking.md",
      "id": 229,
      "releaseDate": "2025-09-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "claude-sonnet-4-5",
        "familyName": "Claude Sonnet 4.5",
        "variantType": "reasoning",
        "snapshotLabel": "thinking",
        "baseFamilyModelKey": "claude-sonnet-4-5",
        "relatedModelKeys": [
          "claude-sonnet-4-5"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {
          "vibeCodeBench": 22.621
        },
        "reasoning": {},
        "multimodalGrounded": {
          "designArenaWebsite": 1208
        },
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "holo2-235b-a22b",
      "canonicalModelKey": "holo2-235b-a22b",
      "model": "Holo2-235B-A22B",
      "creator": "H Company",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/holo2-235b-a22b",
      "markdownUrl": "https://benchlm.ai/md/models/holo2-235b-a22b.md",
      "id": 223,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "holo2",
        "familyName": "Holo2",
        "variantType": "235b-a22b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "holo2-235b-a22b",
        "relatedModelKeys": [
          "holo2-30b-a3b",
          "holo2-8b",
          "holo2-4b"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {
          "screenSpotPro": 70.6
        },
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "holo2-30b-a3b",
      "canonicalModelKey": "holo2-30b-a3b",
      "model": "Holo2-30B-A3B",
      "creator": "H Company",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/holo2-30b-a3b",
      "markdownUrl": "https://benchlm.ai/md/models/holo2-30b-a3b.md",
      "id": 224,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "holo2",
        "familyName": "Holo2",
        "variantType": "30b-a3b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "holo2-235b-a22b",
        "relatedModelKeys": [
          "holo2-235b-a22b",
          "holo2-8b",
          "holo2-4b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {
          "screenSpotPro": 66.1
        },
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "holo2-4b",
      "canonicalModelKey": "holo2-4b",
      "model": "Holo2-4B",
      "creator": "H Company",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/holo2-4b",
      "markdownUrl": "https://benchlm.ai/md/models/holo2-4b.md",
      "id": 226,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "holo2",
        "familyName": "Holo2",
        "variantType": "4b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "holo2-235b-a22b",
        "relatedModelKeys": [
          "holo2-235b-a22b",
          "holo2-30b-a3b",
          "holo2-8b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {
          "screenSpotPro": 57.2
        },
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "holo2-8b",
      "canonicalModelKey": "holo2-8b",
      "model": "Holo2-8B",
      "creator": "H Company",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/holo2-8b",
      "markdownUrl": "https://benchlm.ai/md/models/holo2-8b.md",
      "id": 225,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "holo2",
        "familyName": "Holo2",
        "variantType": "8b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "holo2-235b-a22b",
        "relatedModelKeys": [
          "holo2-235b-a22b",
          "holo2-30b-a3b",
          "holo2-4b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 1,
        "verifiedBenchmarkCount": 1,
        "rankableBenchmarkCount": 1,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {
          "screenSpotPro": 58.9
        },
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "muse-voice-transcribe",
      "canonicalModelKey": "muse-voice-transcribe",
      "model": "Muse Voice Transcribe",
      "creator": "Meta",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/muse-voice-transcribe",
      "markdownUrl": "https://benchlm.ai/md/models/muse-voice-transcribe.md",
      "id": 415,
      "releaseDate": "2026-09-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "muse-voice-transcribe",
        "familyName": "Muse Voice Transcribe",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "muse-voice-transcribe",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "mercury-2-5-preview",
      "canonicalModelKey": "mercury-2-5-preview",
      "model": "Mercury 2.5 Preview",
      "creator": "Inception",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "260K",
      "contextWindowTokens": 260000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/mercury-2-5-preview",
      "markdownUrl": "https://benchlm.ai/md/models/mercury-2-5-preview.md",
      "id": 414,
      "releaseDate": "2026-08-31",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "mercury-2-5",
        "familyName": "Mercury 2.5",
        "variantType": "preview",
        "snapshotLabel": "Preview",
        "baseFamilyModelKey": "mercury-2-5-preview",
        "relatedModelKeys": [
          "mercury-2"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "mercury-2"
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "aaAgenticIndex": 9.46,
          "tau2Bench": 70.8,
          "gdpvalAaNormalized": 10.1,
          "gdpvalAa": 702
        },
        "coding": {
          "aaCodingIndex": 31.11,
          "aaSciCode": 38.7
        },
        "reasoning": {
          "lcr": 40.7,
          "critpt": 0.8
        },
        "multimodalGrounded": {},
        "knowledge": {
          "artificialAnalysis": 21.9,
          "aaGpqaDiamond": 77,
          "aaHle": 17.1,
          "aaOmniscienceIndex": -50.7,
          "omniscienceAccuracy": 21.2,
          "omniscienceHallucinationRate": 91.2
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 69.8
        },
        "math": {}
      }
    },
    {
      "slug": "toast-1",
      "canonicalModelKey": "toast-1",
      "model": "Toast 1",
      "creator": "Mixedbread",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "131K",
      "contextWindowTokens": 131000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/toast-1",
      "markdownUrl": "https://benchlm.ai/md/models/toast-1.md",
      "id": 395,
      "releaseDate": "2026-08-13",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "toast-1",
        "familyName": "Toast 1",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "toast-1",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "gpt-5-6-cyber",
      "canonicalModelKey": "gpt-5-6-cyber",
      "model": "GPT-5.6 Cyber",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": null,
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/gpt-5-6-cyber",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-5-6-cyber.md",
      "id": 390,
      "releaseDate": "2026-08-10",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-5-6",
        "familyName": "GPT-5.6",
        "variantType": "cyber",
        "snapshotLabel": "daybreak-red",
        "baseFamilyModelKey": "gpt-5-6-sol",
        "relatedModelKeys": [
          "gpt-5-6-sol"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {},
        "external": {
          "advancedCyberCompletionRateRed": 95
        }
      }
    },
    {
      "slug": "lfg-3",
      "canonicalModelKey": "lfg-3",
      "model": "LFG-3",
      "creator": "glenn2",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/lfg-3",
      "markdownUrl": "https://benchlm.ai/md/models/lfg-3.md",
      "id": 400,
      "releaseDate": "2026-08-06",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "lfg-3",
        "familyName": "LFG-3",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "lfg-3",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "hark-handoff",
      "canonicalModelKey": "hark-handoff",
      "model": "Hark Handoff",
      "creator": "Hark",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": null,
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/hark-handoff",
      "markdownUrl": "https://benchlm.ai/md/models/hark-handoff.md",
      "id": 384,
      "releaseDate": "2026-08-05",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "hark-handoff",
        "familyName": "Hark Handoff",
        "variantType": "research-preview",
        "snapshotLabel": "Research Preview",
        "baseFamilyModelKey": "hark-handoff",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "macaw-v1",
      "canonicalModelKey": "macaw-v1",
      "model": "Macaw",
      "creator": "Bad Theory Labs",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/macaw-v1",
      "markdownUrl": "https://benchlm.ai/md/models/macaw-v1.md",
      "id": 387,
      "releaseDate": "2026-08-05",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "macaw",
        "familyName": "Macaw",
        "variantType": "v1",
        "snapshotLabel": "v1",
        "baseFamilyModelKey": "macaw-v1",
        "relatedModelKeys": [
          "lfm2-5-2-6b"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "qwen3-7-flash",
      "canonicalModelKey": "qwen3-7-flash",
      "model": "Qwen3.7 Flash",
      "creator": "Alibaba",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/qwen3-7-flash",
      "markdownUrl": "https://benchlm.ai/md/models/qwen3-7-flash.md",
      "id": 296,
      "releaseDate": "2026-07-27",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "qwen3-7-flash",
        "familyName": "Qwen3.7 Flash",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "qwen3-7-flash",
        "relatedModelKeys": [
          "qwen3-7-max",
          "qwen3-7-plus",
          "qwen3-5-flash"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "sakana-fugu-ultra-v1-1",
      "canonicalModelKey": "sakana-fugu-ultra-v1-1",
      "model": "Sakana Fugu-Ultra v1.1",
      "creator": "Sakana AI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/sakana-fugu-ultra-v1-1",
      "markdownUrl": "https://benchlm.ai/md/models/sakana-fugu-ultra-v1-1.md",
      "id": 293,
      "releaseDate": "2026-07-24",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "sakana-fugu",
        "familyName": "Sakana Fugu",
        "variantType": "ultra",
        "snapshotLabel": "Ultra v1.1",
        "baseFamilyModelKey": "sakana-fugu",
        "relatedModelKeys": [
          "sakana-fugu-ultra",
          "sakana-fugu",
          "sakana-fugu-cyber"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "sakana-fugu-ultra"
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "gemini-3-5-flash-cyber",
      "canonicalModelKey": "gemini-3-5-flash-cyber",
      "model": "Gemini 3.5 Flash Cyber",
      "creator": "Google",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": null,
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/gemini-3-5-flash-cyber",
      "markdownUrl": "https://benchlm.ai/md/models/gemini-3-5-flash-cyber.md",
      "id": 291,
      "releaseDate": "2026-07-21",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gemini-3-5-flash",
        "familyName": "Gemini 3.5 Flash",
        "variantType": "cyber",
        "snapshotLabel": "Cyber",
        "baseFamilyModelKey": "gemini-3-5-flash",
        "relatedModelKeys": [
          "gemini-3-5-flash",
          "gemini-3-6-flash"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "cyberGym": 83.2
        },
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "cosmos3-edge",
      "canonicalModelKey": "cosmos3-edge",
      "model": "Cosmos3-Edge",
      "creator": "NVIDIA",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/cosmos3-edge",
      "markdownUrl": "https://benchlm.ai/md/models/cosmos3-edge.md",
      "id": 289,
      "releaseDate": "2026-07-20",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "cosmos3",
        "familyName": "Cosmos 3",
        "variantType": "edge",
        "snapshotLabel": "4B",
        "baseFamilyModelKey": "cosmos3-edge",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "agents-a1-f16-gguf",
      "canonicalModelKey": "agents-a1-f16-gguf",
      "model": "Agents-A1-F16-GGUF",
      "creator": "InternScience",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": 61.2,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 51.33,
        "upper": 71.07
      },
      "rankingEligible": true,
      "overallRank": 59,
      "url": "https://benchlm.ai/models/agents-a1-f16-gguf",
      "markdownUrl": "https://benchlm.ai/md/models/agents-a1-f16-gguf.md",
      "id": 278,
      "releaseDate": "2026-07-02",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "agents-a1",
        "familyName": "Agents-A1",
        "variantType": "f16-gguf",
        "snapshotLabel": null,
        "baseFamilyModelKey": "agents-a1",
        "relatedModelKeys": [
          "agents-a1",
          "agents-a1-fp8",
          "agents-a1-q4-k-m-gguf",
          "agents-a1-q8-0-gguf"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 59,
        "categoryRanks": {
          "agentic": 33
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "agents-a1-fp8",
      "canonicalModelKey": "agents-a1-fp8",
      "model": "Agents-A1-FP8",
      "creator": "InternScience",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": 61.2,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 51.33,
        "upper": 71.07
      },
      "rankingEligible": true,
      "overallRank": 60,
      "url": "https://benchlm.ai/models/agents-a1-fp8",
      "markdownUrl": "https://benchlm.ai/md/models/agents-a1-fp8.md",
      "id": 275,
      "releaseDate": "2026-07-02",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "agents-a1",
        "familyName": "Agents-A1",
        "variantType": "fp8",
        "snapshotLabel": null,
        "baseFamilyModelKey": "agents-a1",
        "relatedModelKeys": [
          "agents-a1",
          "agents-a1-q4-k-m-gguf",
          "agents-a1-q8-0-gguf",
          "agents-a1-f16-gguf"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 60,
        "categoryRanks": {
          "agentic": 34
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "agents-a1-q4-k-m-gguf",
      "canonicalModelKey": "agents-a1-q4-k-m-gguf",
      "model": "Agents-A1-Q4_K_M-GGUF",
      "creator": "InternScience",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": 61.2,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 51.33,
        "upper": 71.07
      },
      "rankingEligible": true,
      "overallRank": 61,
      "url": "https://benchlm.ai/models/agents-a1-q4-k-m-gguf",
      "markdownUrl": "https://benchlm.ai/md/models/agents-a1-q4-k-m-gguf.md",
      "id": 276,
      "releaseDate": "2026-07-02",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "agents-a1",
        "familyName": "Agents-A1",
        "variantType": "q4-k-m-gguf",
        "snapshotLabel": null,
        "baseFamilyModelKey": "agents-a1",
        "relatedModelKeys": [
          "agents-a1",
          "agents-a1-fp8",
          "agents-a1-q8-0-gguf",
          "agents-a1-f16-gguf"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 61,
        "categoryRanks": {
          "agentic": 35
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "agents-a1-q8-0-gguf",
      "canonicalModelKey": "agents-a1-q8-0-gguf",
      "model": "Agents-A1-Q8_0-GGUF",
      "creator": "InternScience",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": 61.2,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 51.33,
        "upper": 71.07
      },
      "rankingEligible": true,
      "overallRank": 62,
      "url": "https://benchlm.ai/models/agents-a1-q8-0-gguf",
      "markdownUrl": "https://benchlm.ai/md/models/agents-a1-q8-0-gguf.md",
      "id": 277,
      "releaseDate": "2026-07-02",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "agents-a1",
        "familyName": "Agents-A1",
        "variantType": "q8-0-gguf",
        "snapshotLabel": null,
        "baseFamilyModelKey": "agents-a1",
        "relatedModelKeys": [
          "agents-a1",
          "agents-a1-fp8",
          "agents-a1-q4-k-m-gguf",
          "agents-a1-f16-gguf"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 62,
        "categoryRanks": {
          "agentic": 36
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "lfg-2",
      "canonicalModelKey": "lfg-2",
      "model": "LFG-2",
      "creator": "glenn2",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/lfg-2",
      "markdownUrl": "https://benchlm.ai/md/models/lfg-2.md",
      "id": 401,
      "releaseDate": "2026-06-30",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "lfg-2",
        "familyName": "LFG-2",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "lfg-2",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "longcat-2-0",
      "canonicalModelKey": "longcat-2-0",
      "model": "LongCat-2.0",
      "creator": "Meituan",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "1M",
      "contextWindowTokens": 1000000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/longcat-2-0",
      "markdownUrl": "https://benchlm.ai/md/models/longcat-2-0.md",
      "id": 272,
      "releaseDate": "2026-06-30",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "longcat-2-0",
        "familyName": "LongCat-2.0",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "longcat-2-0",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "lfm2-5-colbert-350m",
      "canonicalModelKey": "lfm2-5-colbert-350m",
      "model": "LFM2.5-ColBERT-350M",
      "creator": "LiquidAI",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/lfm2-5-colbert-350m",
      "markdownUrl": "https://benchlm.ai/md/models/lfm2-5-colbert-350m.md",
      "id": 261,
      "releaseDate": "2026-06-18",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "lfm2-5-retrievers",
        "familyName": "LFM2.5 Retrievers",
        "variantType": "colbert",
        "snapshotLabel": null,
        "baseFamilyModelKey": "lfm2-5-embedding-350m",
        "relatedModelKeys": [
          "lfm2-5-embedding-350m",
          "lfm2-5-350m"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {
          "nanoBeirMultilingual": 60.5,
          "mkqa11": 69.4
        },
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "lfm2-5-embedding-350m",
      "canonicalModelKey": "lfm2-5-embedding-350m",
      "model": "LFM2.5-Embedding-350M",
      "creator": "LiquidAI",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/lfm2-5-embedding-350m",
      "markdownUrl": "https://benchlm.ai/md/models/lfm2-5-embedding-350m.md",
      "id": 260,
      "releaseDate": "2026-06-18",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "lfm2-5-retrievers",
        "familyName": "LFM2.5 Retrievers",
        "variantType": "embedding",
        "snapshotLabel": null,
        "baseFamilyModelKey": "lfm2-5-embedding-350m",
        "relatedModelKeys": [
          "lfm2-5-colbert-350m",
          "lfm2-5-350m"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {
          "nanoBeirMultilingual": 57.7,
          "mkqa11": 69.1
        },
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "mai-voice-2",
      "canonicalModelKey": "mai-voice-2",
      "model": "MAI-Voice-2",
      "creator": "Microsoft",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/mai-voice-2",
      "markdownUrl": "https://benchlm.ai/md/models/mai-voice-2.md",
      "id": 308,
      "releaseDate": "2026-06-02",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "mai-voice",
        "familyName": "MAI-Voice",
        "variantType": "quality",
        "snapshotLabel": "Voice 2",
        "baseFamilyModelKey": "mai-voice-2",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "holo3-1-0-8b",
      "canonicalModelKey": "holo3-1-0-8b",
      "model": "Holo3.1-0.8B",
      "creator": "H Company",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/holo3-1-0-8b",
      "markdownUrl": "https://benchlm.ai/md/models/holo3-1-0-8b.md",
      "id": 248,
      "releaseDate": "2026-06-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "holo3-1",
        "familyName": "Holo3.1",
        "variantType": "0-8b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "holo3-1-35b-a3b",
        "relatedModelKeys": [
          "holo3-1-35b-a3b",
          "holo3-1-35b-a3b-fp8",
          "holo3-1-35b-a3b-nvfp4",
          "holo3-1-35b-a3b-gguf",
          "holo3-1-9b",
          "holo3-1-4b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "holo3-1-35b-a3b-fp8",
      "canonicalModelKey": "holo3-1-35b-a3b-fp8",
      "model": "Holo3.1-35B-A3B-FP8",
      "creator": "H Company",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/holo3-1-35b-a3b-fp8",
      "markdownUrl": "https://benchlm.ai/md/models/holo3-1-35b-a3b-fp8.md",
      "id": 243,
      "releaseDate": "2026-06-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "holo3-1",
        "familyName": "Holo3.1",
        "variantType": "35b-a3b-fp8",
        "snapshotLabel": null,
        "baseFamilyModelKey": "holo3-1-35b-a3b",
        "relatedModelKeys": [
          "holo3-1-35b-a3b",
          "holo3-1-35b-a3b-nvfp4",
          "holo3-1-35b-a3b-gguf",
          "holo3-1-9b",
          "holo3-1-4b",
          "holo3-1-0-8b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": "holo3-1-35b-a3b"
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "holo3-1-35b-a3b-gguf",
      "canonicalModelKey": "holo3-1-35b-a3b-gguf",
      "model": "Holo3.1-35B-A3B-GGUF",
      "creator": "H Company",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/holo3-1-35b-a3b-gguf",
      "markdownUrl": "https://benchlm.ai/md/models/holo3-1-35b-a3b-gguf.md",
      "id": 245,
      "releaseDate": "2026-06-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "holo3-1",
        "familyName": "Holo3.1",
        "variantType": "35b-a3b-gguf",
        "snapshotLabel": null,
        "baseFamilyModelKey": "holo3-1-35b-a3b",
        "relatedModelKeys": [
          "holo3-1-35b-a3b",
          "holo3-1-35b-a3b-fp8",
          "holo3-1-35b-a3b-nvfp4",
          "holo3-1-9b",
          "holo3-1-4b",
          "holo3-1-0-8b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": "holo3-1-35b-a3b"
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "holo3-1-35b-a3b-nvfp4",
      "canonicalModelKey": "holo3-1-35b-a3b-nvfp4",
      "model": "Holo3.1-35B-A3B-NVFP4",
      "creator": "H Company",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/holo3-1-35b-a3b-nvfp4",
      "markdownUrl": "https://benchlm.ai/md/models/holo3-1-35b-a3b-nvfp4.md",
      "id": 244,
      "releaseDate": "2026-06-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "holo3-1",
        "familyName": "Holo3.1",
        "variantType": "35b-a3b-nvfp4",
        "snapshotLabel": null,
        "baseFamilyModelKey": "holo3-1-35b-a3b",
        "relatedModelKeys": [
          "holo3-1-35b-a3b",
          "holo3-1-35b-a3b-fp8",
          "holo3-1-35b-a3b-gguf",
          "holo3-1-9b",
          "holo3-1-4b",
          "holo3-1-0-8b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": "holo3-1-35b-a3b"
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "lfm2-5-1-2b-jp-202606",
      "canonicalModelKey": "lfm2-5-1-2b-jp-202606",
      "model": "LFM2.5-1.2B-JP-202606",
      "creator": "LiquidAI",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/lfm2-5-1-2b-jp-202606",
      "markdownUrl": "https://benchlm.ai/md/models/lfm2-5-1-2b-jp-202606.md",
      "id": 254,
      "releaseDate": "2026-05-26",
      "market": "Japan",
      "isRegional": true,
      "family": {
        "familyKey": "lfm2-5-1-2b",
        "familyName": "LFM2.5-1.2B",
        "variantType": "jp-202606",
        "snapshotLabel": null,
        "baseFamilyModelKey": "lfm2-5-1-2b-instruct",
        "relatedModelKeys": [
          "lfm2-5-1-2b-instruct",
          "lfm2-5-1-2b-thinking"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "hy-mt2-1-8b",
      "canonicalModelKey": "hy-mt2-1-8b",
      "model": "Hy-MT2-1.8B",
      "creator": "Tencent",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "8K",
      "contextWindowTokens": 8000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/hy-mt2-1-8b",
      "markdownUrl": "https://benchlm.ai/md/models/hy-mt2-1-8b.md",
      "id": 404,
      "releaseDate": "2026-05-21",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "hy-mt2",
        "familyName": "Hy-MT2",
        "variantType": "1-8b",
        "snapshotLabel": "1.8B",
        "baseFamilyModelKey": "hy-mt2-30b-a3b",
        "relatedModelKeys": [
          "hy-mt2-30b-a3b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "hy-mt2-30b-a3b",
      "canonicalModelKey": "hy-mt2-30b-a3b",
      "model": "Hy-MT2-30B-A3B",
      "creator": "Tencent",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "8K",
      "contextWindowTokens": 8000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/hy-mt2-30b-a3b",
      "markdownUrl": "https://benchlm.ai/md/models/hy-mt2-30b-a3b.md",
      "id": 403,
      "releaseDate": "2026-05-21",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "hy-mt2",
        "familyName": "Hy-MT2",
        "variantType": "30b-a3b",
        "snapshotLabel": "30B-A3B",
        "baseFamilyModelKey": "hy-mt2-30b-a3b",
        "relatedModelKeys": [
          "hy-mt2-1-8b"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "grok-build-0-1",
      "canonicalModelKey": "grok-build-0-1",
      "model": "Grok Build 0.1",
      "creator": "xAI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/grok-build-0-1",
      "markdownUrl": "https://benchlm.ai/md/models/grok-build-0-1.md",
      "id": 173,
      "releaseDate": "2026-05-20",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "grok-build",
        "familyName": "Grok Build",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "grok-build-0-1",
        "relatedModelKeys": [
          "grok-code-fast-1"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {
          "agentic": 135,
          "coding": 121
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "gertLabs": 49.15
        },
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "cartesia-sonic-3-5",
      "canonicalModelKey": "cartesia-sonic-3-5",
      "model": "Cartesia Sonic 3.5",
      "creator": "Cartesia",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/cartesia-sonic-3-5",
      "markdownUrl": "https://benchlm.ai/md/models/cartesia-sonic-3-5.md",
      "id": 311,
      "releaseDate": "2026-05-04",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "cartesia-sonic",
        "familyName": "Cartesia Sonic",
        "variantType": "v3.5",
        "snapshotLabel": "Stable alias",
        "baseFamilyModelKey": "cartesia-sonic-3-5",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "hy-mt1-5-1-8b-1-25bit",
      "canonicalModelKey": "hy-mt1-5-1-8b-1-25bit",
      "model": "Hy-MT1.5-1.8B-1.25bit",
      "creator": "Tencent Hunyuan",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/hy-mt1-5-1-8b-1-25bit",
      "markdownUrl": "https://benchlm.ai/md/models/hy-mt1-5-1-8b-1-25bit.md",
      "id": 237,
      "releaseDate": "2026-04-29",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "hy-mt1-5-1-8b",
        "familyName": "Hy-MT1.5-1.8B",
        "variantType": "1.25bit",
        "snapshotLabel": null,
        "baseFamilyModelKey": "hy-mt1-5-1-8b-1-25bit",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "lfg-1",
      "canonicalModelKey": "lfg-1",
      "model": "LFG-1",
      "creator": "glenn2",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/lfg-1",
      "markdownUrl": "https://benchlm.ai/md/models/lfg-1.md",
      "id": 324,
      "releaseDate": "2026-04-20",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "lfg-1",
        "familyName": "LFG-1",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "lfg-1",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "grok-tts",
      "canonicalModelKey": "grok-tts",
      "model": "Grok TTS",
      "creator": "xAI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/grok-tts",
      "markdownUrl": "https://benchlm.ai/md/models/grok-tts.md",
      "id": 309,
      "releaseDate": "2026-04-17",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "grok-voice",
        "familyName": "Grok Voice",
        "variantType": "tts",
        "snapshotLabel": "TTS API",
        "baseFamilyModelKey": "grok-tts",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "ternary-bonsai-1-7b",
      "canonicalModelKey": "ternary-bonsai-1-7b",
      "model": "Ternary Bonsai 1.7B",
      "creator": "Prism ML",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/ternary-bonsai-1-7b",
      "markdownUrl": "https://benchlm.ai/md/models/ternary-bonsai-1-7b.md",
      "id": 109,
      "releaseDate": "2026-04-16",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "ternary-bonsai",
        "familyName": "Ternary Bonsai",
        "variantType": "1-7b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "ternary-bonsai-8b",
        "relatedModelKeys": [
          "ternary-bonsai-8b",
          "ternary-bonsai-4b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "ternary-bonsai-4b",
      "canonicalModelKey": "ternary-bonsai-4b",
      "model": "Ternary Bonsai 4B",
      "creator": "Prism ML",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/ternary-bonsai-4b",
      "markdownUrl": "https://benchlm.ai/md/models/ternary-bonsai-4b.md",
      "id": 120,
      "releaseDate": "2026-04-16",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "ternary-bonsai",
        "familyName": "Ternary Bonsai",
        "variantType": "4b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "ternary-bonsai-8b",
        "relatedModelKeys": [
          "ternary-bonsai-8b",
          "ternary-bonsai-1-7b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "ternary-bonsai-8b",
      "canonicalModelKey": "ternary-bonsai-8b",
      "model": "Ternary Bonsai 8B",
      "creator": "Prism ML",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "64K",
      "contextWindowTokens": 64000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/ternary-bonsai-8b",
      "markdownUrl": "https://benchlm.ai/md/models/ternary-bonsai-8b.md",
      "id": 80,
      "releaseDate": "2026-04-16",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "ternary-bonsai",
        "familyName": "Ternary Bonsai",
        "variantType": "8b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "ternary-bonsai-8b",
        "relatedModelKeys": [
          "ternary-bonsai-4b",
          "ternary-bonsai-1-7b"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "gemini-3-1-flash-tts-preview",
      "canonicalModelKey": "gemini-3-1-flash-tts-preview",
      "model": "Gemini 3.1 Flash TTS Preview",
      "creator": "Google",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/gemini-3-1-flash-tts-preview",
      "markdownUrl": "https://benchlm.ai/md/models/gemini-3-1-flash-tts-preview.md",
      "id": 313,
      "releaseDate": "2026-04-13",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gemini-tts",
        "familyName": "Gemini TTS",
        "variantType": "flash-preview",
        "snapshotLabel": "3.1 Flash Preview",
        "baseFamilyModelKey": "gemini-2-5-flash-tts-preview",
        "relatedModelKeys": [
          "gemini-2-5-flash-tts-preview",
          "gemini-2-5-pro-tts-preview"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "gemma-4-e2b",
      "canonicalModelKey": "gemma-4-e2b",
      "model": "Gemma 4 E2B",
      "creator": "Google",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 42.33,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 30.82,
        "upper": 53.85
      },
      "rankingEligible": true,
      "overallRank": 184,
      "url": "https://benchlm.ai/models/gemma-4-e2b",
      "markdownUrl": "https://benchlm.ai/md/models/gemma-4-e2b.md",
      "id": 161,
      "releaseDate": "2026-04-02",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gemma-4",
        "familyName": "Gemma 4",
        "variantType": "e2b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gemma-4-31b",
        "relatedModelKeys": [
          "gemma-4-31b",
          "gemma-4-26b-a4b",
          "gemma-4-e4b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 184,
        "categoryRanks": {
          "coding": 125
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 20.8,
          "gdpvalAaNormalized": 0,
          "gdpvalAa": 89
        },
        "coding": {
          "aaSciCode": 20.9,
          "aaCodingIndex": 7.23
        },
        "reasoning": {
          "lcr": 17,
          "critpt": 0
        },
        "multimodalGrounded": {
          "aaMmmuPro": 44.6
        },
        "knowledge": {
          "gpqa": 43.4,
          "mmluPro": 60,
          "artificialAnalysis": 9.53,
          "aaGpqaDiamond": 43.3,
          "aaHle": 4.8,
          "aaOmniscienceIndex": -23.6,
          "omniscienceAccuracy": 6.6,
          "omniscienceHallucinationRate": 32.4
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 38
        },
        "math": {}
      }
    },
    {
      "slug": "gemma-4-e4b",
      "canonicalModelKey": "gemma-4-e4b",
      "model": "Gemma 4 E4B",
      "creator": "Google",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 43.37,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 31.86,
        "upper": 54.89
      },
      "rankingEligible": true,
      "overallRank": 178,
      "url": "https://benchlm.ai/models/gemma-4-e4b",
      "markdownUrl": "https://benchlm.ai/md/models/gemma-4-e4b.md",
      "id": 139,
      "releaseDate": "2026-04-02",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gemma-4",
        "familyName": "Gemma 4",
        "variantType": "e4b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gemma-4-31b",
        "relatedModelKeys": [
          "gemma-4-31b",
          "gemma-4-26b-a4b",
          "gemma-4-e2b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 178,
        "categoryRanks": {
          "coding": 119
        },
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 20.8,
          "gdpvalAaNormalized": 0,
          "gdpvalAa": 231
        },
        "coding": {
          "aaSciCode": 24.4,
          "aaCodingIndex": 9.39
        },
        "reasoning": {
          "lcr": 33,
          "critpt": 0.6
        },
        "multimodalGrounded": {
          "aaMmmuPro": 51.4
        },
        "knowledge": {
          "gpqa": 58.6,
          "mmluPro": 69.4,
          "artificialAnalysis": 12.18,
          "aaGpqaDiamond": 57.6,
          "aaHle": 3.8,
          "aaOmniscienceIndex": -19.7,
          "omniscienceAccuracy": 8.6,
          "omniscienceHallucinationRate": 30.9
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 44.2
        },
        "math": {}
      }
    },
    {
      "slug": "bonsai-1-7b",
      "canonicalModelKey": "bonsai-1-7b",
      "model": "1-bit Bonsai 1.7B",
      "creator": "Prism ML",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/bonsai-1-7b",
      "markdownUrl": "https://benchlm.ai/md/models/bonsai-1-7b.md",
      "id": 160,
      "releaseDate": "2026-03-31",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "bonsai",
        "familyName": "1-bit Bonsai",
        "variantType": "1-7b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "bonsai-8b",
        "relatedModelKeys": [
          "bonsai-8b",
          "bonsai-4b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "bonsai-4b",
      "canonicalModelKey": "bonsai-4b",
      "model": "1-bit Bonsai 4B",
      "creator": "Prism ML",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/bonsai-4b",
      "markdownUrl": "https://benchlm.ai/md/models/bonsai-4b.md",
      "id": 148,
      "releaseDate": "2026-03-31",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "bonsai",
        "familyName": "1-bit Bonsai",
        "variantType": "4b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "bonsai-8b",
        "relatedModelKeys": [
          "bonsai-8b",
          "bonsai-1-7b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "bonsai-8b",
      "canonicalModelKey": "bonsai-8b",
      "model": "1-bit Bonsai 8B",
      "creator": "Prism ML",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "64K",
      "contextWindowTokens": 64000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/bonsai-8b",
      "markdownUrl": "https://benchlm.ai/md/models/bonsai-8b.md",
      "id": 126,
      "releaseDate": "2026-03-31",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "bonsai",
        "familyName": "1-bit Bonsai",
        "variantType": "8b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "bonsai-8b",
        "relatedModelKeys": [
          "bonsai-4b",
          "bonsai-1-7b"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "lfm2-5-350m",
      "canonicalModelKey": "lfm2-5-350m",
      "model": "LFM2.5-350M",
      "creator": "LiquidAI",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/lfm2-5-350m",
      "markdownUrl": "https://benchlm.ai/md/models/lfm2-5-350m.md",
      "id": 159,
      "releaseDate": "2026-03-31",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "lfm2-5-350m",
        "familyName": "LFM2.5-350M",
        "variantType": "instruct",
        "snapshotLabel": null,
        "baseFamilyModelKey": "lfm2-5-350m",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "leanstral",
      "canonicalModelKey": "leanstral",
      "model": "Leanstral",
      "creator": "Mistral",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/leanstral",
      "markdownUrl": "https://benchlm.ai/md/models/leanstral.md",
      "id": 203,
      "releaseDate": "2026-03-15",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "leanstral",
        "familyName": "Leanstral",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "leanstral",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "grok-4-20-multi-agent-beta",
      "canonicalModelKey": "grok-4-20-multi-agent-beta",
      "model": "Grok 4.20 Multi-agent",
      "creator": "xAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "2M",
      "contextWindowTokens": 2000000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/grok-4-20-multi-agent-beta",
      "markdownUrl": "https://benchlm.ai/md/models/grok-4-20-multi-agent-beta.md",
      "id": 8,
      "releaseDate": "2026-03-10",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "grok-4-20",
        "familyName": "Grok 4.20",
        "variantType": "multi-agent",
        "snapshotLabel": null,
        "baseFamilyModelKey": "grok-4-20-beta",
        "relatedModelKeys": [
          "grok-4-20-beta"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "sarvam-105b",
      "canonicalModelKey": "sarvam-105b",
      "model": "Sarvam 105B",
      "creator": "Sarvam",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 43.26,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 31.75,
        "upper": 54.78
      },
      "rankingEligible": true,
      "overallRank": 179,
      "url": "https://benchlm.ai/models/sarvam-105b",
      "markdownUrl": "https://benchlm.ai/md/models/sarvam-105b.md",
      "id": 115,
      "releaseDate": "2026-03-06",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "sarvam-105b",
        "familyName": "Sarvam 105B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "sarvam-105b",
        "relatedModelKeys": [
          "sarvam-30b"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "sarvam-m"
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 179,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 46.8
        },
        "coding": {
          "aaSciCode": 26.4
        },
        "reasoning": {
          "lcr": 0,
          "critpt": 0
        },
        "multimodalGrounded": {},
        "knowledge": {
          "artificialAnalysis": 11.9,
          "aaGpqaDiamond": 73.8,
          "aaHle": 11,
          "aaOmniscienceIndex": -59.4,
          "omniscienceAccuracy": 17.6,
          "omniscienceHallucinationRate": 93.4
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 34.4
        },
        "math": {}
      }
    },
    {
      "slug": "sarvam-30b",
      "canonicalModelKey": "sarvam-30b",
      "model": "Sarvam 30B",
      "creator": "Sarvam",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "64K",
      "contextWindowTokens": 64000,
      "displayScore": 41.1,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 29.58,
        "upper": 52.61
      },
      "rankingEligible": true,
      "overallRank": 190,
      "url": "https://benchlm.ai/models/sarvam-30b",
      "markdownUrl": "https://benchlm.ai/md/models/sarvam-30b.md",
      "id": 136,
      "releaseDate": "2026-03-06",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "sarvam-30b",
        "familyName": "Sarvam 30B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "sarvam-30b",
        "relatedModelKeys": [
          "sarvam-105b"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": "sarvam-m"
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 190,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 34.5
        },
        "coding": {
          "aaSciCode": 19.2
        },
        "reasoning": {
          "lcr": 0,
          "critpt": 0.3
        },
        "multimodalGrounded": {},
        "knowledge": {
          "artificialAnalysis": 6.39,
          "aaGpqaDiamond": 63.3,
          "aaHle": 7.5,
          "aaOmniscienceIndex": -71.5,
          "omniscienceAccuracy": 12.6,
          "omniscienceHallucinationRate": 96.3
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 26.5
        },
        "math": {}
      }
    },
    {
      "slug": "mistral-medium-3",
      "canonicalModelKey": "mistral-medium-3",
      "model": "Mistral Medium 3",
      "creator": "Mistral",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 43.49,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 31.98,
        "upper": 55
      },
      "rankingEligible": true,
      "overallRank": 177,
      "url": "https://benchlm.ai/models/mistral-medium-3",
      "markdownUrl": "https://benchlm.ai/md/models/mistral-medium-3.md",
      "id": 106,
      "releaseDate": "2026-02-20",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "mistral-medium-3",
        "familyName": "Mistral Medium 3",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "mistral-medium-3",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 177,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 24.3
        },
        "coding": {
          "aaSciCode": 33.1
        },
        "reasoning": {
          "lcr": 31.7,
          "critpt": 0
        },
        "multimodalGrounded": {
          "aaMmmuPro": 53,
          "designArenaWebsite": 1097
        },
        "knowledge": {
          "artificialAnalysis": 12.47,
          "aaGpqaDiamond": 57.8,
          "aaHle": 4.1,
          "aaOmniscienceIndex": -31.4,
          "omniscienceAccuracy": 18.3,
          "omniscienceHallucinationRate": 60.9
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 39.3
        },
        "math": {}
      }
    },
    {
      "slug": "mistral-small-4-reasoning",
      "canonicalModelKey": "mistral-small-4-reasoning",
      "model": "Mistral Small 4 (Reasoning)",
      "creator": "Mistral",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/mistral-small-4-reasoning",
      "markdownUrl": "https://benchlm.ai/md/models/mistral-small-4-reasoning.md",
      "id": 49,
      "releaseDate": "2026-02-20",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "mistral-small-4",
        "familyName": "Mistral Small 4",
        "variantType": "reasoning",
        "snapshotLabel": null,
        "baseFamilyModelKey": "mistral-small-4",
        "relatedModelKeys": [
          "mistral-small-4"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "aaAgenticIndex": 4.63,
          "tau2Bench": 41.2,
          "gdpvalAaNormalized": 4.5,
          "gdpvalAa": 590
        },
        "coding": {
          "aaCodingIndex": 26.64,
          "aaSciCode": 38
        },
        "reasoning": {
          "lcr": 47.3,
          "critpt": 0.3
        },
        "multimodalGrounded": {
          "aaMmmuPro": 56.8
        },
        "knowledge": {
          "artificialAnalysis": 19.71,
          "aaGpqaDiamond": 76.9,
          "aaHle": 9.9,
          "aaOmniscienceIndex": -30.4,
          "omniscienceAccuracy": 21.7,
          "omniscienceHallucinationRate": 66.5
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 48.2
        },
        "math": {}
      }
    },
    {
      "slug": "elevenlabs-eleven-v3",
      "canonicalModelKey": "elevenlabs-eleven-v3",
      "model": "ElevenLabs Eleven v3",
      "creator": "ElevenLabs",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/elevenlabs-eleven-v3",
      "markdownUrl": "https://benchlm.ai/md/models/elevenlabs-eleven-v3.md",
      "id": 317,
      "releaseDate": "2026-02-02",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "elevenlabs-eleven",
        "familyName": "ElevenLabs Eleven",
        "variantType": "v3",
        "snapshotLabel": "Generally available",
        "baseFamilyModelKey": "elevenlabs-eleven-v3",
        "relatedModelKeys": [
          "elevenlabs-eleven-v3-conversational"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "elevenlabs-eleven-v3-conversational",
      "canonicalModelKey": "elevenlabs-eleven-v3-conversational",
      "model": "ElevenLabs Eleven v3 Conversational",
      "creator": "ElevenLabs",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/elevenlabs-eleven-v3-conversational",
      "markdownUrl": "https://benchlm.ai/md/models/elevenlabs-eleven-v3-conversational.md",
      "id": 412,
      "releaseDate": "2026-02-02",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "elevenlabs-eleven",
        "familyName": "ElevenLabs Eleven",
        "variantType": "v3-conversational",
        "snapshotLabel": "Conversational",
        "baseFamilyModelKey": "elevenlabs-eleven-v3",
        "relatedModelKeys": [
          "elevenlabs-eleven-v3"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "gpt-realtime-1-5",
      "canonicalModelKey": "gpt-realtime-1-5",
      "model": "GPT Realtime 1.5",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/gpt-realtime-1-5",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-realtime-1-5.md",
      "id": 361,
      "releaseDate": "2026-02-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-realtime-1-5",
        "familyName": "GPT Realtime 1.5",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-realtime-1-5",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "inworld-tts-1-5-max",
      "canonicalModelKey": "inworld-tts-1-5-max",
      "model": "Inworld TTS-1.5 Max",
      "creator": "Inworld",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/inworld-tts-1-5-max",
      "markdownUrl": "https://benchlm.ai/md/models/inworld-tts-1-5-max.md",
      "id": 318,
      "releaseDate": "2026-01-21",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "inworld-tts",
        "familyName": "Inworld TTS",
        "variantType": "max",
        "snapshotLabel": "1.5 Max",
        "baseFamilyModelKey": "inworld-tts-1-5-max",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "gemini-2-5-flash-native-audio-preview-12-2025",
      "canonicalModelKey": "gemini-2-5-flash-native-audio-preview-12-2025",
      "model": "Gemini 2.5 Flash Native Audio Preview (12-2025)",
      "creator": "Google",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/gemini-2-5-flash-native-audio-preview-12-2025",
      "markdownUrl": "https://benchlm.ai/md/models/gemini-2-5-flash-native-audio-preview-12-2025.md",
      "id": 355,
      "releaseDate": "2025-12-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gemini-2-5-flash-native-audio-preview-12-2025",
        "familyName": "Gemini 2.5 Flash Native Audio Preview (12-2025)",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gemini-2-5-flash-native-audio-preview-12-2025",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "granite-4-0-1b",
      "canonicalModelKey": "granite-4-0-1b",
      "model": "Granite-4.0-1B",
      "creator": "IBM",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/granite-4-0-1b",
      "markdownUrl": "https://benchlm.ai/md/models/granite-4-0-1b.md",
      "id": 157,
      "releaseDate": "2025-10-28",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "granite-4-0-1b",
        "familyName": "Granite 4.0 1B",
        "variantType": "dense",
        "snapshotLabel": null,
        "baseFamilyModelKey": "granite-4-0-h-1b",
        "relatedModelKeys": [
          "granite-4-0-h-1b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 22.8
        },
        "coding": {
          "aaSciCode": 8.7
        },
        "reasoning": {
          "lcr": 6,
          "critpt": 0
        },
        "multimodalGrounded": {},
        "knowledge": {
          "artificialAnalysis": 1.63,
          "aaGpqaDiamond": 28.1,
          "aaHle": 4.8,
          "aaOmniscienceIndex": -81.6,
          "omniscienceAccuracy": 6.2,
          "omniscienceHallucinationRate": 93.5
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 20.5
        },
        "math": {}
      }
    },
    {
      "slug": "granite-4-0-350m",
      "canonicalModelKey": "granite-4-0-350m",
      "model": "Granite-4.0-350M",
      "creator": "IBM",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": 39,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 27.49,
        "upper": 50.51
      },
      "rankingEligible": true,
      "overallRank": 201,
      "url": "https://benchlm.ai/models/granite-4-0-350m",
      "markdownUrl": "https://benchlm.ai/md/models/granite-4-0-350m.md",
      "id": 171,
      "releaseDate": "2025-10-28",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "granite-4-0-350m",
        "familyName": "Granite 4.0 350M",
        "variantType": "dense",
        "snapshotLabel": null,
        "baseFamilyModelKey": "granite-4-0-350m",
        "relatedModelKeys": [
          "granite-4-0-h-350m"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 201,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 13.2
        },
        "coding": {
          "aaSciCode": 0.9
        },
        "reasoning": {
          "lcr": 0,
          "critpt": 0
        },
        "multimodalGrounded": {},
        "knowledge": {
          "artificialAnalysis": 1,
          "aaGpqaDiamond": 26.1,
          "aaHle": 5.5,
          "aaOmniscienceIndex": -69.3,
          "omniscienceAccuracy": 3.9,
          "omniscienceHallucinationRate": 76.1
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 15.9
        },
        "math": {}
      }
    },
    {
      "slug": "granite-4-0-h-1b",
      "canonicalModelKey": "granite-4-0-h-1b",
      "model": "Granite-4.0-H-1B",
      "creator": "IBM",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/granite-4-0-h-1b",
      "markdownUrl": "https://benchlm.ai/md/models/granite-4-0-h-1b.md",
      "id": 152,
      "releaseDate": "2025-10-28",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "granite-4-0-1b",
        "familyName": "Granite 4.0 1B",
        "variantType": "hybrid",
        "snapshotLabel": null,
        "baseFamilyModelKey": "granite-4-0-h-1b",
        "relatedModelKeys": [
          "granite-4-0-1b"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 19.6
        },
        "coding": {
          "aaSciCode": 8.2
        },
        "reasoning": {
          "lcr": 6,
          "critpt": 0
        },
        "multimodalGrounded": {},
        "knowledge": {
          "artificialAnalysis": 2.25,
          "aaGpqaDiamond": 26.3,
          "aaHle": 5,
          "aaOmniscienceIndex": -72.3,
          "omniscienceAccuracy": 5.2,
          "omniscienceHallucinationRate": 81.7
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 26.2
        },
        "math": {}
      }
    },
    {
      "slug": "granite-4-0-h-350m",
      "canonicalModelKey": "granite-4-0-h-350m",
      "model": "Granite-4.0-H-350M",
      "creator": "IBM",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": 39,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 27.49,
        "upper": 50.51
      },
      "rankingEligible": true,
      "overallRank": 202,
      "url": "https://benchlm.ai/models/granite-4-0-h-350m",
      "markdownUrl": "https://benchlm.ai/md/models/granite-4-0-h-350m.md",
      "id": 172,
      "releaseDate": "2025-10-28",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "granite-4-0-350m",
        "familyName": "Granite 4.0 350M",
        "variantType": "hybrid",
        "snapshotLabel": null,
        "baseFamilyModelKey": "granite-4-0-350m",
        "relatedModelKeys": [
          "granite-4-0-350m"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 202,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 14.6
        },
        "coding": {
          "aaSciCode": 1.7
        },
        "reasoning": {
          "lcr": 0,
          "critpt": 0
        },
        "multimodalGrounded": {},
        "knowledge": {
          "artificialAnalysis": 1,
          "aaGpqaDiamond": 25.7,
          "aaHle": 6.4,
          "aaOmniscienceIndex": -80.9,
          "omniscienceAccuracy": 3.8,
          "omniscienceHallucinationRate": 88.1
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 17.6
        },
        "math": {}
      }
    },
    {
      "slug": "gpt-5-nano",
      "canonicalModelKey": "gpt-5-nano",
      "model": "GPT-5 nano",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "400K",
      "contextWindowTokens": 400000,
      "displayScore": 46.53,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 35.01,
        "upper": 58.04
      },
      "rankingEligible": true,
      "overallRank": 159,
      "url": "https://benchlm.ai/models/gpt-5-nano",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-5-nano.md",
      "id": 194,
      "releaseDate": "2025-08-07",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-5",
        "familyName": "GPT-5",
        "variantType": "nano",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-5-high",
        "relatedModelKeys": [
          "gpt-5-high",
          "gpt-5-medium",
          "gpt-5-mini"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 159,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 18,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {},
        "korean": {}
      }
    },
    {
      "slug": "audio-flamingo-3-7b",
      "canonicalModelKey": "audio-flamingo-3-7b",
      "model": "Audio Flamingo 3 7B",
      "creator": "NVIDIA",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/audio-flamingo-3-7b",
      "markdownUrl": "https://benchlm.ai/md/models/audio-flamingo-3-7b.md",
      "id": 364,
      "releaseDate": "2025-07-10",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "audio-flamingo-3-7b",
        "familyName": "Audio Flamingo 3 7B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "audio-flamingo-3-7b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "gemini-2-5-flash-tts-preview",
      "canonicalModelKey": "gemini-2-5-flash-tts-preview",
      "model": "Gemini 2.5 Flash TTS Preview",
      "creator": "Google",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/gemini-2-5-flash-tts-preview",
      "markdownUrl": "https://benchlm.ai/md/models/gemini-2-5-flash-tts-preview.md",
      "id": 315,
      "releaseDate": "2025-05-20",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gemini-tts",
        "familyName": "Gemini TTS",
        "variantType": "flash-preview",
        "snapshotLabel": "2.5 Flash Preview",
        "baseFamilyModelKey": "gemini-2-5-flash-tts-preview",
        "relatedModelKeys": [
          "gemini-2-5-pro-tts-preview",
          "gemini-3-1-flash-tts-preview"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "gemini-2-5-pro-tts-preview",
      "canonicalModelKey": "gemini-2-5-pro-tts-preview",
      "model": "Gemini 2.5 Pro TTS Preview",
      "creator": "Google",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/gemini-2-5-pro-tts-preview",
      "markdownUrl": "https://benchlm.ai/md/models/gemini-2-5-pro-tts-preview.md",
      "id": 310,
      "releaseDate": "2025-05-20",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gemini-tts",
        "familyName": "Gemini TTS",
        "variantType": "pro-preview",
        "snapshotLabel": "2.5 Pro Preview",
        "baseFamilyModelKey": "gemini-2-5-flash-tts-preview",
        "relatedModelKeys": [
          "gemini-2-5-flash-tts-preview",
          "gemini-3-1-flash-tts-preview"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "kimi-audio-7b",
      "canonicalModelKey": "kimi-audio-7b",
      "model": "Kimi-Audio 7B",
      "creator": "Moonshot AI",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/kimi-audio-7b",
      "markdownUrl": "https://benchlm.ai/md/models/kimi-audio-7b.md",
      "id": 325,
      "releaseDate": "2025-04-25",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "kimi-audio-7b",
        "familyName": "Kimi-Audio 7B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "kimi-audio-7b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "qwen2-5-omni-7b",
      "canonicalModelKey": "qwen2-5-omni-7b",
      "model": "Qwen2.5-Omni 7B",
      "creator": "Alibaba",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/qwen2-5-omni-7b",
      "markdownUrl": "https://benchlm.ai/md/models/qwen2-5-omni-7b.md",
      "id": 363,
      "releaseDate": "2025-03-26",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "qwen2-5-omni-7b",
        "familyName": "Qwen2.5-Omni 7B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "qwen2-5-omni-7b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "gpt-4o-mini-tts",
      "canonicalModelKey": "gpt-4o-mini-tts",
      "model": "GPT-4o mini TTS",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "2K",
      "contextWindowTokens": 2000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/gpt-4o-mini-tts",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-4o-mini-tts.md",
      "id": 314,
      "releaseDate": "2025-03-20",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-4o-mini-tts",
        "familyName": "GPT-4o mini TTS",
        "variantType": "base",
        "snapshotLabel": "Current alias",
        "baseFamilyModelKey": "gpt-4o-mini-tts",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "phi-4-multimodal-instruct",
      "canonicalModelKey": "phi-4-multimodal-instruct",
      "model": "Phi-4 Multimodal Instruct",
      "creator": "Microsoft",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/phi-4-multimodal-instruct",
      "markdownUrl": "https://benchlm.ai/md/models/phi-4-multimodal-instruct.md",
      "id": 333,
      "releaseDate": "2025-03-03",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "phi-4-multimodal-instruct",
        "familyName": "Phi-4 Multimodal Instruct",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "phi-4-multimodal-instruct",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {
          "aaSciCode": 11
        },
        "reasoning": {},
        "multimodalGrounded": {
          "aaMmmuPro": 14.5
        },
        "knowledge": {
          "artificialAnalysis": 4.2,
          "aaGpqaDiamond": 31.5,
          "aaHle": 5
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "baichuan-audio",
      "canonicalModelKey": "baichuan-audio",
      "model": "Baichuan-Audio",
      "creator": "Baichuan",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/baichuan-audio",
      "markdownUrl": "https://benchlm.ai/md/models/baichuan-audio.md",
      "id": 330,
      "releaseDate": "2025-02-24",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "baichuan-audio",
        "familyName": "Baichuan-Audio",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "baichuan-audio",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "grok-3-mini",
      "canonicalModelKey": "grok-3-mini",
      "model": "Grok 3 Mini",
      "creator": "xAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/grok-3-mini",
      "markdownUrl": "https://benchlm.ai/md/models/grok-3-mini.md",
      "id": 134,
      "releaseDate": "2025-02-19",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "grok-3-mini",
        "familyName": "Grok 3 Mini",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "grok-3-mini",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "step-audio",
      "canonicalModelKey": "step-audio",
      "model": "Step-Audio",
      "creator": "StepFun",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/step-audio",
      "markdownUrl": "https://benchlm.ai/md/models/step-audio.md",
      "id": 340,
      "releaseDate": "2025-02-17",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "step-audio",
        "familyName": "Step-Audio",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "step-audio",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "ola-omni",
      "canonicalModelKey": "ola-omni",
      "model": "Ola",
      "creator": "Ola-Omni",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/ola-omni",
      "markdownUrl": "https://benchlm.ai/md/models/ola-omni.md",
      "id": 334,
      "releaseDate": "2025-02-06",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "ola-omni",
        "familyName": "Ola",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "ola-omni",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "baichuan-omni-1-5",
      "canonicalModelKey": "baichuan-omni-1-5",
      "model": "Baichuan-Omni 1.5",
      "creator": "Baichuan",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/baichuan-omni-1-5",
      "markdownUrl": "https://benchlm.ai/md/models/baichuan-omni-1-5.md",
      "id": 328,
      "releaseDate": "2025-01-26",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "baichuan-omni-1-5",
        "familyName": "Baichuan-Omni 1.5",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "baichuan-omni-1-5",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "minicpm-o-2-6",
      "canonicalModelKey": "minicpm-o-2-6",
      "model": "MiniCPM-o 2.6",
      "creator": "OpenBMB",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/minicpm-o-2-6",
      "markdownUrl": "https://benchlm.ai/md/models/minicpm-o-2-6.md",
      "id": 329,
      "releaseDate": "2025-01-24",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "minicpm-o-2-6",
        "familyName": "MiniCPM-o 2.6",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "minicpm-o-2-6",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "deepseek-r1-distill-qwen-32b",
      "canonicalModelKey": "deepseek-r1-distill-qwen-32b",
      "model": "DeepSeek R1 Distill Qwen 32B",
      "creator": "DeepSeek",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": 42.89,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 31.38,
        "upper": 54.41
      },
      "rankingEligible": true,
      "overallRank": 182,
      "url": "https://benchlm.ai/models/deepseek-r1-distill-qwen-32b",
      "markdownUrl": "https://benchlm.ai/md/models/deepseek-r1-distill-qwen-32b.md",
      "id": 217,
      "releaseDate": "2025-01-20",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "deepseek-r1-distill",
        "familyName": "DeepSeek R1 Distill",
        "variantType": "qwen-32b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "deepseek-r1-distill-qwen-32b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 182,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {
          "aaSciCode": 37.6
        },
        "reasoning": {
          "lcr": 8.3
        },
        "multimodalGrounded": {},
        "knowledge": {
          "artificialAnalysis": 10.96,
          "aaGpqaDiamond": 61.5,
          "aaHle": 4.6
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 22.9
        },
        "math": {}
      }
    },
    {
      "slug": "vita-1-5",
      "canonicalModelKey": "vita-1-5",
      "model": "VITA 1.5",
      "creator": "VITA-MLLM",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/vita-1-5",
      "markdownUrl": "https://benchlm.ai/md/models/vita-1-5.md",
      "id": 332,
      "releaseDate": "2025-01-03",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "vita-1-5",
        "familyName": "VITA 1.5",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "vita-1-5",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "slam-omni",
      "canonicalModelKey": "slam-omni",
      "model": "SLAM-Omni",
      "creator": "X-LANCE",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/slam-omni",
      "markdownUrl": "https://benchlm.ai/md/models/slam-omni.md",
      "id": 347,
      "releaseDate": "2024-12-20",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "slam-omni",
        "familyName": "SLAM-Omni",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "slam-omni",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "meralion-audiollm",
      "canonicalModelKey": "meralion-audiollm",
      "model": "MERaLiON-AudioLLM",
      "creator": "AI Singapore",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/meralion-audiollm",
      "markdownUrl": "https://benchlm.ai/md/models/meralion-audiollm.md",
      "id": 331,
      "releaseDate": "2024-12-13",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "meralion-audiollm",
        "familyName": "MERaLiON-AudioLLM",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "meralion-audiollm",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "lyra-base",
      "canonicalModelKey": "lyra-base",
      "model": "Lyra Base",
      "creator": "DVL Lab",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/lyra-base",
      "markdownUrl": "https://benchlm.ai/md/models/lyra-base.md",
      "id": 335,
      "releaseDate": "2024-12-12",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "lyra-base",
        "familyName": "Lyra Base",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "lyra-base",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "lyra-mini",
      "canonicalModelKey": "lyra-mini",
      "model": "Lyra Mini",
      "creator": "DVL Lab",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/lyra-mini",
      "markdownUrl": "https://benchlm.ai/md/models/lyra-mini.md",
      "id": 343,
      "releaseDate": "2024-12-12",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "lyra-mini",
        "familyName": "Lyra Mini",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "lyra-mini",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "glm-4-voice",
      "canonicalModelKey": "glm-4-voice",
      "model": "GLM-4-Voice",
      "creator": "Zhipu AI",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/glm-4-voice",
      "markdownUrl": "https://benchlm.ai/md/models/glm-4-voice.md",
      "id": 338,
      "releaseDate": "2024-12-03",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "glm-4-voice",
        "familyName": "GLM-4-Voice",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "glm-4-voice",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "o1-pro",
      "canonicalModelKey": "o1-pro",
      "model": "o1-pro",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "200K",
      "contextWindowTokens": 200000,
      "displayScore": 46.12,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": "estimated",
      "scoreInterval90": {
        "lower": 34.61,
        "upper": 57.63
      },
      "rankingEligible": true,
      "overallRank": 162,
      "url": "https://benchlm.ai/models/o1-pro",
      "markdownUrl": "https://benchlm.ai/md/models/o1-pro.md",
      "id": 141,
      "releaseDate": "2024-12-01",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "o1",
        "familyName": "o1",
        "variantType": "pro",
        "snapshotLabel": null,
        "baseFamilyModelKey": "o1",
        "relatedModelKeys": [
          "o1",
          "o1-preview"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": 162,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {
          "gpqa": 79,
          "artificialAnalysis": 19.12
        },
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "ichigo",
      "canonicalModelKey": "ichigo",
      "model": "Ichigo",
      "creator": "Jan",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/ichigo",
      "markdownUrl": "https://benchlm.ai/md/models/ichigo.md",
      "id": 342,
      "releaseDate": "2024-10-20",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "ichigo",
        "familyName": "Ichigo",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "ichigo",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "mini-omni2",
      "canonicalModelKey": "mini-omni2",
      "model": "Mini-Omni2",
      "creator": "gpt-omni",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/mini-omni2",
      "markdownUrl": "https://benchlm.ai/md/models/mini-omni2.md",
      "id": 348,
      "releaseDate": "2024-10-15",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "mini-omni2",
        "familyName": "Mini-Omni2",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "mini-omni2",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "diva-llama-3-v0-8b",
      "canonicalModelKey": "diva-llama-3-v0-8b",
      "model": "DiVA Llama 3 8B",
      "creator": "DiVA authors",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/diva-llama-3-v0-8b",
      "markdownUrl": "https://benchlm.ai/md/models/diva-llama-3-v0-8b.md",
      "id": 337,
      "releaseDate": "2024-10-03",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "diva-llama-3-v0-8b",
        "familyName": "DiVA Llama 3 8B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "diva-llama-3-v0-8b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "moshi-7b",
      "canonicalModelKey": "moshi-7b",
      "model": "Moshi 7B",
      "creator": "Kyutai",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/moshi-7b",
      "markdownUrl": "https://benchlm.ai/md/models/moshi-7b.md",
      "id": 371,
      "releaseDate": "2024-09-17",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "moshi-7b",
        "familyName": "Moshi 7B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "moshi-7b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "llama-omni",
      "canonicalModelKey": "llama-omni",
      "model": "LLaMA-Omni",
      "creator": "ICTNLP",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/llama-omni",
      "markdownUrl": "https://benchlm.ai/md/models/llama-omni.md",
      "id": 345,
      "releaseDate": "2024-09-10",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "llama-omni",
        "familyName": "LLaMA-Omni",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "llama-omni",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "mini-omni",
      "canonicalModelKey": "mini-omni",
      "model": "Mini-Omni 0.5B",
      "creator": "gpt-omni",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/mini-omni",
      "markdownUrl": "https://benchlm.ai/md/models/mini-omni.md",
      "id": 349,
      "releaseDate": "2024-08-29",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "mini-omni",
        "familyName": "Mini-Omni 0.5B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "mini-omni",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "vita-1-0",
      "canonicalModelKey": "vita-1-0",
      "model": "VITA 1.0",
      "creator": "VITA-MLLM",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/vita-1-0",
      "markdownUrl": "https://benchlm.ai/md/models/vita-1-0.md",
      "id": 346,
      "releaseDate": "2024-08-09",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "vita-1-0",
        "familyName": "VITA 1.0",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "vita-1-0",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "qwen2-audio-7b-instruct",
      "canonicalModelKey": "qwen2-audio-7b-instruct",
      "model": "Qwen2-Audio 7B Instruct",
      "creator": "Alibaba",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/qwen2-audio-7b-instruct",
      "markdownUrl": "https://benchlm.ai/md/models/qwen2-audio-7b-instruct.md",
      "id": 339,
      "releaseDate": "2024-07-15",
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "qwen2-audio-7b-instruct",
        "familyName": "Qwen2-Audio 7B Instruct",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "qwen2-audio-7b-instruct",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "skt-ax",
      "canonicalModelKey": "skt-ax",
      "model": "A.X series",
      "creator": "SK Telecom",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "64K",
      "contextWindowTokens": 64000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/skt-ax",
      "markdownUrl": "https://benchlm.ai/md/models/skt-ax.md",
      "id": 212,
      "releaseDate": null,
      "market": "Korea",
      "isRegional": true,
      "family": {
        "familyKey": "skt-ax",
        "familyName": "A.X",
        "variantType": null,
        "snapshotLabel": null,
        "baseFamilyModelKey": "skt-ax",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {},
        "korean": {}
      }
    },
    {
      "slug": "amazon-nova-2-sonic",
      "canonicalModelKey": "amazon-nova-2-sonic",
      "model": "Amazon Nova 2 Sonic",
      "creator": "Amazon",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/amazon-nova-2-sonic",
      "markdownUrl": "https://benchlm.ai/md/models/amazon-nova-2-sonic.md",
      "id": 358,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "amazon-nova-2-sonic",
        "familyName": "Amazon Nova 2 Sonic",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "amazon-nova-2-sonic",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "bland-speech-v3",
      "canonicalModelKey": "bland-speech-v3",
      "model": "Bland Speech v3",
      "creator": "Bland AI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/bland-speech-v3",
      "markdownUrl": "https://benchlm.ai/md/models/bland-speech-v3.md",
      "id": 307,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "bland-speech",
        "familyName": "Bland Speech",
        "variantType": "v3",
        "snapshotLabel": "v3",
        "baseFamilyModelKey": "bland-speech-v3",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "blsp-7b",
      "canonicalModelKey": "blsp-7b",
      "model": "BLSP 7B",
      "creator": "BLSP authors",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/blsp-7b",
      "markdownUrl": "https://benchlm.ai/md/models/blsp-7b.md",
      "id": 373,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "blsp-7b",
        "familyName": "BLSP 7B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "blsp-7b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "cartesia-ink-whisper",
      "canonicalModelKey": "cartesia-ink-whisper",
      "model": "Cartesia Ink-Whisper",
      "creator": "Cartesia",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/cartesia-ink-whisper",
      "markdownUrl": "https://benchlm.ai/md/models/cartesia-ink-whisper.md",
      "id": 377,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "cartesia-ink-whisper",
        "familyName": "Cartesia Ink-Whisper",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "cartesia-ink-whisper",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "cartesia-sonic-3",
      "canonicalModelKey": "cartesia-sonic-3",
      "model": "Cartesia Sonic 3",
      "creator": "Cartesia",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/cartesia-sonic-3",
      "markdownUrl": "https://benchlm.ai/md/models/cartesia-sonic-3.md",
      "id": 378,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "cartesia-sonic-3",
        "familyName": "Cartesia Sonic 3",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "cartesia-sonic-3",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "cohere-transcribe-03-2026",
      "canonicalModelKey": "cohere-transcribe-03-2026",
      "model": "Cohere Transcribe 03-2026",
      "creator": "Cohere",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/cohere-transcribe-03-2026",
      "markdownUrl": "https://benchlm.ai/md/models/cohere-transcribe-03-2026.md",
      "id": 375,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "cohere-transcribe-03-2026",
        "familyName": "Cohere Transcribe 03-2026",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "cohere-transcribe-03-2026",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "dbrx-instruct",
      "canonicalModelKey": "dbrx-instruct",
      "model": "DBRX Instruct",
      "creator": "Databricks",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/dbrx-instruct",
      "markdownUrl": "https://benchlm.ai/md/models/dbrx-instruct.md",
      "id": 154,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "dbrx",
        "familyName": "DBRX",
        "variantType": "instruct",
        "snapshotLabel": null,
        "baseFamilyModelKey": "dbrx-instruct",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "deepgram-aura-2",
      "canonicalModelKey": "deepgram-aura-2",
      "model": "Deepgram Aura-2",
      "creator": "Deepgram",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/deepgram-aura-2",
      "markdownUrl": "https://benchlm.ai/md/models/deepgram-aura-2.md",
      "id": 382,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "deepgram-aura-2",
        "familyName": "Deepgram Aura-2",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "deepgram-aura-2",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "deepgram-nova-3",
      "canonicalModelKey": "deepgram-nova-3",
      "model": "Deepgram Nova-3",
      "creator": "Deepgram",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/deepgram-nova-3",
      "markdownUrl": "https://benchlm.ai/md/models/deepgram-nova-3.md",
      "id": 379,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "deepgram-nova-3",
        "familyName": "Deepgram Nova-3",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "deepgram-nova-3",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "dna-1-0",
      "canonicalModelKey": "dna-1-0",
      "model": "DNA 1.0 8B",
      "creator": "Community",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/dna-1-0",
      "markdownUrl": "https://benchlm.ai/md/models/dna-1-0.md",
      "id": 215,
      "releaseDate": null,
      "market": "Korea",
      "isRegional": true,
      "family": {
        "familyKey": "dna-1-0",
        "familyName": "DNA 1.0",
        "variantType": null,
        "snapshotLabel": null,
        "baseFamilyModelKey": "dna-1-0",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {},
        "korean": {}
      }
    },
    {
      "slug": "exaone-4-0-1-2b",
      "canonicalModelKey": "exaone-4-0-1-2b",
      "model": "Exaone 4.0 1.2B",
      "creator": "LG AI Research",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/exaone-4-0-1-2b",
      "markdownUrl": "https://benchlm.ai/md/models/exaone-4-0-1-2b.md",
      "id": 205,
      "releaseDate": null,
      "market": "Korea",
      "isRegional": true,
      "family": {
        "familyKey": "exaone-4",
        "familyName": "Exaone 4.0",
        "variantType": "1-2b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "exaone-4-0-32b",
        "relatedModelKeys": [
          "exaone-4-0-32b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 20.5
        },
        "coding": {
          "aaSciCode": 7.4
        },
        "reasoning": {
          "lcr": 0,
          "critpt": 0
        },
        "multimodalGrounded": {},
        "knowledge": {
          "artificialAnalysis": 2.37,
          "aaGpqaDiamond": 42.4,
          "aaHle": 5.7,
          "aaOmniscienceIndex": -82.1,
          "omniscienceAccuracy": 5,
          "omniscienceHallucinationRate": 91.7
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 25.3
        },
        "math": {},
        "korean": {}
      }
    },
    {
      "slug": "gemini-3-1-flash-live-preview",
      "canonicalModelKey": "gemini-3-1-flash-live-preview",
      "model": "Gemini 3.1 Flash Live Preview",
      "creator": "Google",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/gemini-3-1-flash-live-preview",
      "markdownUrl": "https://benchlm.ai/md/models/gemini-3-1-flash-live-preview.md",
      "id": 354,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gemini-3-1-flash-live-preview",
        "familyName": "Gemini 3.1 Flash Live Preview",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gemini-3-1-flash-live-preview",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "glm-realtime-air",
      "canonicalModelKey": "glm-realtime-air",
      "model": "GLM Realtime Air",
      "creator": "Zhipu AI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/glm-realtime-air",
      "markdownUrl": "https://benchlm.ai/md/models/glm-realtime-air.md",
      "id": 360,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "glm-realtime-air",
        "familyName": "GLM Realtime Air",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "glm-realtime-air",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "glm-realtime-flash",
      "canonicalModelKey": "glm-realtime-flash",
      "model": "GLM Realtime Flash",
      "creator": "Zhipu AI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/glm-realtime-flash",
      "markdownUrl": "https://benchlm.ai/md/models/glm-realtime-flash.md",
      "id": 359,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "glm-realtime-flash",
        "familyName": "GLM Realtime Flash",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "glm-realtime-flash",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "gpt-realtime",
      "canonicalModelKey": "gpt-realtime",
      "model": "GPT Realtime",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/gpt-realtime",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-realtime.md",
      "id": 356,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-realtime",
        "familyName": "GPT Realtime",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-realtime",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "gpt-realtime-2",
      "canonicalModelKey": "gpt-realtime-2",
      "model": "GPT Realtime 2",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/gpt-realtime-2",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-realtime-2.md",
      "id": 357,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-realtime-2",
        "familyName": "GPT Realtime 2",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-realtime-2",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "gpt-realtime-mini",
      "canonicalModelKey": "gpt-realtime-mini",
      "model": "GPT Realtime mini",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/gpt-realtime-mini",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-realtime-mini.md",
      "id": 362,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-realtime-mini",
        "familyName": "GPT Realtime mini",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-realtime-mini",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "gpt-4o-audio",
      "canonicalModelKey": "gpt-4o-audio",
      "model": "GPT-4o Audio",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/gpt-4o-audio",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-4o-audio.md",
      "id": 321,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-4o-audio",
        "familyName": "GPT-4o Audio",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-4o-audio",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "gpt-4o-mini-audio",
      "canonicalModelKey": "gpt-4o-mini-audio",
      "model": "GPT-4o mini Audio",
      "creator": "OpenAI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/gpt-4o-mini-audio",
      "markdownUrl": "https://benchlm.ai/md/models/gpt-4o-mini-audio.md",
      "id": 322,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "gpt-4o-mini-audio",
        "familyName": "GPT-4o mini Audio",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "gpt-4o-mini-audio",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "grok-realtime",
      "canonicalModelKey": "grok-realtime",
      "model": "Grok Realtime",
      "creator": "xAI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/grok-realtime",
      "markdownUrl": "https://benchlm.ai/md/models/grok-realtime.md",
      "id": 352,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "grok-realtime",
        "familyName": "Grok Realtime",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "grok-realtime",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "grok-voice-think-fast-1-0",
      "canonicalModelKey": "grok-voice-think-fast-1-0",
      "model": "Grok Voice Think Fast 1.0",
      "creator": "xAI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/grok-voice-think-fast-1-0",
      "markdownUrl": "https://benchlm.ai/md/models/grok-voice-think-fast-1-0.md",
      "id": 353,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "grok-voice-think-fast-1-0",
        "familyName": "Grok Voice Think Fast 1.0",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "grok-voice-think-fast-1-0",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "grok-voice-think-fast-2-0",
      "canonicalModelKey": "grok-voice-think-fast-2-0",
      "model": "Grok Voice Think Fast 2.0",
      "creator": "xAI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/grok-voice-think-fast-2-0",
      "markdownUrl": "https://benchlm.ai/md/models/grok-voice-think-fast-2-0.md",
      "id": 350,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "grok-voice-think-fast-2-0",
        "familyName": "Grok Voice Think Fast 2.0",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "grok-voice-think-fast-2-0",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "holotron-12b",
      "canonicalModelKey": "holotron-12b",
      "model": "Holotron-12B",
      "creator": "H Company",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/holotron-12b",
      "markdownUrl": "https://benchlm.ai/md/models/holotron-12b.md",
      "id": 222,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "holotron",
        "familyName": "Holotron",
        "variantType": "12b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "holotron-12b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "hyperclova-x-dash",
      "canonicalModelKey": "hyperclova-x-dash",
      "model": "HyperClova X Dash",
      "creator": "Naver Cloud",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/hyperclova-x-dash",
      "markdownUrl": "https://benchlm.ai/md/models/hyperclova-x-dash.md",
      "id": 207,
      "releaseDate": null,
      "market": "Korea",
      "isRegional": true,
      "family": {
        "familyKey": "hyperclova",
        "familyName": "HyperClova X",
        "variantType": null,
        "snapshotLabel": null,
        "baseFamilyModelKey": "hyperclova-x-think",
        "relatedModelKeys": [
          "hyperclova-x-think",
          "hyperclova-x-seed-8b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {},
        "korean": {}
      }
    },
    {
      "slug": "hyperclova-x-think",
      "canonicalModelKey": "hyperclova-x-think",
      "model": "HyperClova X Think 32B",
      "creator": "Naver Cloud",
      "sourceType": "Open Weight",
      "reasoningType": "Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/hyperclova-x-think",
      "markdownUrl": "https://benchlm.ai/md/models/hyperclova-x-think.md",
      "id": 206,
      "releaseDate": null,
      "market": "Korea",
      "isRegional": true,
      "family": {
        "familyKey": "hyperclova",
        "familyName": "HyperClova X",
        "variantType": null,
        "snapshotLabel": null,
        "baseFamilyModelKey": "hyperclova-x-think",
        "relatedModelKeys": [
          "hyperclova-x-dash",
          "hyperclova-x-seed-8b"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {},
        "korean": {}
      }
    },
    {
      "slug": "k-exaone",
      "canonicalModelKey": "k-exaone",
      "model": "K-Exaone",
      "creator": "LG AI Research",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "256K",
      "contextWindowTokens": 256000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/k-exaone",
      "markdownUrl": "https://benchlm.ai/md/models/k-exaone.md",
      "id": 131,
      "releaseDate": null,
      "market": "Korea",
      "isRegional": true,
      "family": {
        "familyKey": "exaone-frontier",
        "familyName": "K-Exaone",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "k-exaone",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 74.3,
          "gdpvalAaNormalized": 4.6,
          "gdpvalAa": 592
        },
        "coding": {
          "aaSciCode": 35.6,
          "aaCodingIndex": 32.11
        },
        "reasoning": {
          "lcr": 59.3,
          "critpt": 1.1
        },
        "multimodalGrounded": {},
        "knowledge": {
          "artificialAnalysis": 22.47,
          "aaGpqaDiamond": 78.3,
          "aaHle": 13.9,
          "aaOmniscienceIndex": -58,
          "omniscienceAccuracy": 16.4,
          "omniscienceHallucinationRate": 88.8
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 64.7
        },
        "math": {},
        "korean": {}
      }
    },
    {
      "slug": "kalpa-tts-beta-v0-1",
      "canonicalModelKey": "kalpa-tts-beta-v0-1",
      "model": "Kalpa TTS Beta v0.1",
      "creator": "Kalpa Labs",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/kalpa-tts-beta-v0-1",
      "markdownUrl": "https://benchlm.ai/md/models/kalpa-tts-beta-v0-1.md",
      "id": 418,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "kalpa-tts",
        "familyName": "Kalpa TTS",
        "variantType": "beta-v0-1",
        "snapshotLabel": "Beta v0.1",
        "baseFamilyModelKey": "kalpa-tts-beta-v0-1",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "kanana-essence",
      "canonicalModelKey": "kanana-essence",
      "model": "Kanana Essence",
      "creator": "Kakao",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "64K",
      "contextWindowTokens": 64000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/kanana-essence",
      "markdownUrl": "https://benchlm.ai/md/models/kanana-essence.md",
      "id": 210,
      "releaseDate": null,
      "market": "Korea",
      "isRegional": true,
      "family": {
        "familyKey": "kanana",
        "familyName": "Kanana",
        "variantType": null,
        "snapshotLabel": null,
        "baseFamilyModelKey": "kanana-flag",
        "relatedModelKeys": [
          "kanana-flag",
          "kanana-nano"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {},
        "korean": {}
      }
    },
    {
      "slug": "kanana-flag",
      "canonicalModelKey": "kanana-flag",
      "model": "Kanana Flag",
      "creator": "Kakao",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "64K",
      "contextWindowTokens": 64000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/kanana-flag",
      "markdownUrl": "https://benchlm.ai/md/models/kanana-flag.md",
      "id": 209,
      "releaseDate": null,
      "market": "Korea",
      "isRegional": true,
      "family": {
        "familyKey": "kanana",
        "familyName": "Kanana",
        "variantType": null,
        "snapshotLabel": null,
        "baseFamilyModelKey": "kanana-flag",
        "relatedModelKeys": [
          "kanana-essence",
          "kanana-nano"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {},
        "korean": {}
      }
    },
    {
      "slug": "kanana-nano",
      "canonicalModelKey": "kanana-nano",
      "model": "Kanana Nano",
      "creator": "Kakao",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "64K",
      "contextWindowTokens": 64000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/kanana-nano",
      "markdownUrl": "https://benchlm.ai/md/models/kanana-nano.md",
      "id": 211,
      "releaseDate": null,
      "market": "Korea",
      "isRegional": true,
      "family": {
        "familyKey": "kanana",
        "familyName": "Kanana",
        "variantType": null,
        "snapshotLabel": null,
        "baseFamilyModelKey": "kanana-flag",
        "relatedModelKeys": [
          "kanana-flag",
          "kanana-essence"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {},
        "korean": {}
      }
    },
    {
      "slug": "kokoro-82m",
      "canonicalModelKey": "kokoro-82m",
      "model": "Kokoro 82M",
      "creator": "hexgrad",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/kokoro-82m",
      "markdownUrl": "https://benchlm.ai/md/models/kokoro-82m.md",
      "id": 380,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "kokoro-82m",
        "familyName": "Kokoro 82M",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "kokoro-82m",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "lightning-v3-1-pro",
      "canonicalModelKey": "lightning-v3-1-pro",
      "model": "Lightning v3.1 Pro",
      "creator": "Smallest AI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/lightning-v3-1-pro",
      "markdownUrl": "https://benchlm.ai/md/models/lightning-v3-1-pro.md",
      "id": 316,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "smallest-lightning",
        "familyName": "Smallest AI Lightning",
        "variantType": "pro",
        "snapshotLabel": "v3.1 Pro",
        "baseFamilyModelKey": "lightning-v3-1-pro",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "mair-hub-0-5b-omni",
      "canonicalModelKey": "mair-hub-0-5b-omni",
      "model": "Mair-hub 0.5B Omni",
      "creator": "Mair Hub",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/mair-hub-0-5b-omni",
      "markdownUrl": "https://benchlm.ai/md/models/mair-hub-0-5b-omni.md",
      "id": 344,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "mair-hub-0-5b-omni",
        "familyName": "Mair-hub 0.5B Omni",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "mair-hub-0-5b-omni",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "megrez-3b-omni",
      "canonicalModelKey": "megrez-3b-omni",
      "model": "Megrez-3B-Omni",
      "creator": "Infinigence AI",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/megrez-3b-omni",
      "markdownUrl": "https://benchlm.ai/md/models/megrez-3b-omni.md",
      "id": 341,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "megrez-3b-omni",
        "familyName": "Megrez-3B-Omni",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "megrez-3b-omni",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "minimax-speech-02-hd",
      "canonicalModelKey": "minimax-speech-02-hd",
      "model": "MiniMax Speech-02 HD",
      "creator": "MiniMax",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/minimax-speech-02-hd",
      "markdownUrl": "https://benchlm.ai/md/models/minimax-speech-02-hd.md",
      "id": 312,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "minimax-speech",
        "familyName": "MiniMax Speech",
        "variantType": "hd",
        "snapshotLabel": "Speech-02 HD",
        "baseFamilyModelKey": "minimax-speech-02-hd",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "next-gpt-7b",
      "canonicalModelKey": "next-gpt-7b",
      "model": "NExT-GPT 7B",
      "creator": "NExT-GPT authors",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/next-gpt-7b",
      "markdownUrl": "https://benchlm.ai/md/models/next-gpt-7b.md",
      "id": 366,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "next-gpt-7b",
        "familyName": "NExT-GPT 7B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "next-gpt-7b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "orion-mistral-24b",
      "canonicalModelKey": "orion-mistral-24b",
      "model": "OriOn-Mistral-24B",
      "creator": "LightOn",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "344K",
      "contextWindowTokens": 344000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/orion-mistral-24b",
      "markdownUrl": "https://benchlm.ai/md/models/orion-mistral-24b.md",
      "id": 221,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "orion",
        "familyName": "OriOn",
        "variantType": "mistral-24b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "orion-qwen-32b",
        "relatedModelKeys": [
          "orion-qwen-32b"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "orion-qwen-32b",
      "canonicalModelKey": "orion-qwen-32b",
      "model": "OriOn-Qwen-32B",
      "creator": "LightOn",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "262K",
      "contextWindowTokens": 262000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/orion-qwen-32b",
      "markdownUrl": "https://benchlm.ai/md/models/orion-qwen-32b.md",
      "id": 220,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "orion",
        "familyName": "OriOn",
        "variantType": "qwen-32b",
        "snapshotLabel": null,
        "baseFamilyModelKey": "orion-qwen-32b",
        "relatedModelKeys": [
          "orion-mistral-24b"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "pandagpt-7b",
      "canonicalModelKey": "pandagpt-7b",
      "model": "PandaGPT 7B",
      "creator": "PandaGPT authors",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/pandagpt-7b",
      "markdownUrl": "https://benchlm.ai/md/models/pandagpt-7b.md",
      "id": 365,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "pandagpt-7b",
        "familyName": "PandaGPT 7B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "pandagpt-7b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "parakeet-ctc-1-1b",
      "canonicalModelKey": "parakeet-ctc-1-1b",
      "model": "Parakeet CTC 1.1B",
      "creator": "NVIDIA",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/parakeet-ctc-1-1b",
      "markdownUrl": "https://benchlm.ai/md/models/parakeet-ctc-1-1b.md",
      "id": 383,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "parakeet-ctc-1-1b",
        "familyName": "Parakeet CTC 1.1B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "parakeet-ctc-1-1b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "pharia-1-llm-7b-control",
      "canonicalModelKey": "pharia-1-llm-7b-control",
      "model": "Pharia-1-LLM-7B-control",
      "creator": "Aleph Alpha",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "8K",
      "contextWindowTokens": 8000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/pharia-1-llm-7b-control",
      "markdownUrl": "https://benchlm.ai/md/models/pharia-1-llm-7b-control.md",
      "id": 218,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "pharia-1-llm-7b",
        "familyName": "Pharia-1-LLM-7B",
        "variantType": "control",
        "snapshotLabel": null,
        "baseFamilyModelKey": "pharia-1-llm-7b-control",
        "relatedModelKeys": [
          "pharia-1-llm-7b-control-aligned"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "pharia-1-llm-7b-control-aligned",
      "canonicalModelKey": "pharia-1-llm-7b-control-aligned",
      "model": "Pharia-1-LLM-7B-control-aligned",
      "creator": "Aleph Alpha",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "8K",
      "contextWindowTokens": 8000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/pharia-1-llm-7b-control-aligned",
      "markdownUrl": "https://benchlm.ai/md/models/pharia-1-llm-7b-control-aligned.md",
      "id": 219,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "pharia-1-llm-7b",
        "familyName": "Pharia-1-LLM-7B",
        "variantType": "control-aligned",
        "snapshotLabel": null,
        "baseFamilyModelKey": "pharia-1-llm-7b-control",
        "relatedModelKeys": [
          "pharia-1-llm-7b-control"
        ],
        "isCanonicalFamilyEntry": false,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "qwen-audio-7b",
      "canonicalModelKey": "qwen-audio-7b",
      "model": "Qwen-Audio 7B",
      "creator": "Alibaba",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/qwen-audio-7b",
      "markdownUrl": "https://benchlm.ai/md/models/qwen-audio-7b.md",
      "id": 367,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "qwen-audio-7b",
        "familyName": "Qwen-Audio 7B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "qwen-audio-7b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "qwen-audio-chat-7b",
      "canonicalModelKey": "qwen-audio-chat-7b",
      "model": "Qwen-Audio-Chat 7B",
      "creator": "Alibaba",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/qwen-audio-chat-7b",
      "markdownUrl": "https://benchlm.ai/md/models/qwen-audio-chat-7b.md",
      "id": 369,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "qwen-audio-chat-7b",
        "familyName": "Qwen-Audio-Chat 7B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "qwen-audio-chat-7b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "salmonn-13b",
      "canonicalModelKey": "salmonn-13b",
      "model": "SALMONN 13B",
      "creator": "SALMONN authors",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/salmonn-13b",
      "markdownUrl": "https://benchlm.ai/md/models/salmonn-13b.md",
      "id": 372,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "salmonn-13b",
        "familyName": "SALMONN 13B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "salmonn-13b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "salmonn-7b",
      "canonicalModelKey": "salmonn-7b",
      "model": "SALMONN 7B",
      "creator": "SALMONN authors",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/salmonn-7b",
      "markdownUrl": "https://benchlm.ai/md/models/salmonn-7b.md",
      "id": 368,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "salmonn-7b",
        "familyName": "SALMONN 7B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "salmonn-7b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "scribe-v2-2-realtime",
      "canonicalModelKey": "scribe-v2-2-realtime",
      "model": "Scribe v2.2 Realtime",
      "creator": "ElevenLabs",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/scribe-v2-2-realtime",
      "markdownUrl": "https://benchlm.ai/md/models/scribe-v2-2-realtime.md",
      "id": 376,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "scribe-v2-2-realtime",
        "familyName": "Scribe v2.2 Realtime",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "scribe-v2-2-realtime",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "solar-pro-2",
      "canonicalModelKey": "solar-pro-2",
      "model": "Solar Pro 2",
      "creator": "Upstage",
      "sourceType": "Proprietary",
      "reasoningType": "Reasoning",
      "contextWindow": "128K",
      "contextWindowTokens": 128000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/solar-pro-2",
      "markdownUrl": "https://benchlm.ai/md/models/solar-pro-2.md",
      "id": 208,
      "releaseDate": null,
      "market": "Korea",
      "isRegional": true,
      "family": {
        "familyKey": "solar",
        "familyName": "Solar",
        "variantType": null,
        "snapshotLabel": null,
        "baseFamilyModelKey": "solar-pro-2",
        "relatedModelKeys": [
          "solar-wbl"
        ],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 31.9
        },
        "coding": {
          "aaSciCode": 24.8
        },
        "reasoning": {
          "lcr": 0,
          "critpt": 0
        },
        "multimodalGrounded": {},
        "knowledge": {
          "artificialAnalysis": 7.57,
          "aaGpqaDiamond": 56.1,
          "aaHle": 3.7,
          "aaOmniscienceIndex": -62,
          "omniscienceAccuracy": 16.1,
          "omniscienceHallucinationRate": 93
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 33.7
        },
        "math": {},
        "korean": {}
      }
    },
    {
      "slug": "speechgpt-7b",
      "canonicalModelKey": "speechgpt-7b",
      "model": "SpeechGPT 7B",
      "creator": "OpenMOSS",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/speechgpt-7b",
      "markdownUrl": "https://benchlm.ai/md/models/speechgpt-7b.md",
      "id": 370,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "speechgpt-7b",
        "familyName": "SpeechGPT 7B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "speechgpt-7b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "step-audio-chat-130b",
      "canonicalModelKey": "step-audio-chat-130b",
      "model": "Step-Audio-Chat 130B",
      "creator": "StepFun",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/step-audio-chat-130b",
      "markdownUrl": "https://benchlm.ai/md/models/step-audio-chat-130b.md",
      "id": 374,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "step-audio-chat-130b",
        "familyName": "Step-Audio-Chat 130B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "step-audio-chat-130b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "thunder-llm",
      "canonicalModelKey": "thunder-llm",
      "model": "Thunder-LLM 8B",
      "creator": "Academic",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "32K",
      "contextWindowTokens": 32000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/thunder-llm",
      "markdownUrl": "https://benchlm.ai/md/models/thunder-llm.md",
      "id": 214,
      "releaseDate": null,
      "market": "Korea",
      "isRegional": true,
      "family": {
        "familyKey": "thunder-llm",
        "familyName": "Thunder-LLM",
        "variantType": null,
        "snapshotLabel": null,
        "baseFamilyModelKey": "thunder-llm",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {},
        "korean": {}
      }
    },
    {
      "slug": "ultravox-glm-4p6",
      "canonicalModelKey": "ultravox-glm-4p6",
      "model": "Ultravox GLM-4P6",
      "creator": "Fixie AI",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/ultravox-glm-4p6",
      "markdownUrl": "https://benchlm.ai/md/models/ultravox-glm-4p6.md",
      "id": 320,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "ultravox-glm-4p6",
        "familyName": "Ultravox GLM-4P6",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "ultravox-glm-4p6",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "ultravox-glm-4p7",
      "canonicalModelKey": "ultravox-glm-4p7",
      "model": "Ultravox GLM-4P7",
      "creator": "Fixie AI",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/ultravox-glm-4p7",
      "markdownUrl": "https://benchlm.ai/md/models/ultravox-glm-4p7.md",
      "id": 319,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "ultravox-glm-4p7",
        "familyName": "Ultravox GLM-4P7",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "ultravox-glm-4p7",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "ultravox-v0-4-1-llama-3-1-8b",
      "canonicalModelKey": "ultravox-v0-4-1-llama-3-1-8b",
      "model": "Ultravox v0.4.1 Llama 3.1 8B",
      "creator": "Fixie AI",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/ultravox-v0-4-1-llama-3-1-8b",
      "markdownUrl": "https://benchlm.ai/md/models/ultravox-v0-4-1-llama-3-1-8b.md",
      "id": 327,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "ultravox-v0-4-1-llama-3-1-8b",
        "familyName": "Ultravox v0.4.1 Llama 3.1 8B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "ultravox-v0-4-1-llama-3-1-8b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "ultravox-v0-5-llama-3-1-8b",
      "canonicalModelKey": "ultravox-v0-5-llama-3-1-8b",
      "model": "Ultravox v0.5 Llama 3.1 8B",
      "creator": "Fixie AI",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/ultravox-v0-5-llama-3-1-8b",
      "markdownUrl": "https://benchlm.ai/md/models/ultravox-v0-5-llama-3-1-8b.md",
      "id": 326,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "ultravox-v0-5-llama-3-1-8b",
        "familyName": "Ultravox v0.5 Llama 3.1 8B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "ultravox-v0-5-llama-3-1-8b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "ultravox-v0-5-llama-3-2-1b",
      "canonicalModelKey": "ultravox-v0-5-llama-3-2-1b",
      "model": "Ultravox v0.5 Llama 3.2 1B",
      "creator": "Fixie AI",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/ultravox-v0-5-llama-3-2-1b",
      "markdownUrl": "https://benchlm.ai/md/models/ultravox-v0-5-llama-3-2-1b.md",
      "id": 336,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "ultravox-v0-5-llama-3-2-1b",
        "familyName": "Ultravox v0.5 Llama 3.2 1B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "ultravox-v0-5-llama-3-2-1b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "ultravox-v0-6-llama-3-3-70b",
      "canonicalModelKey": "ultravox-v0-6-llama-3-3-70b",
      "model": "Ultravox v0.6 Llama 3.3 70B",
      "creator": "Fixie AI",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/ultravox-v0-6-llama-3-3-70b",
      "markdownUrl": "https://benchlm.ai/md/models/ultravox-v0-6-llama-3-3-70b.md",
      "id": 323,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "ultravox-v0-6-llama-3-3-70b",
        "familyName": "Ultravox v0.6 Llama 3.3 70B",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "ultravox-v0-6-llama-3-3-70b",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {
          "tau2Bench": 26.6,
          "gdpvalAaNormalized": 0,
          "gdpvalAa": 98
        },
        "coding": {
          "aaCodingIndex": 11.93,
          "aaSciCode": 26
        },
        "reasoning": {
          "lcr": 15.7,
          "critpt": 0
        },
        "multimodalGrounded": {},
        "knowledge": {
          "artificialAnalysis": 9.26,
          "aaGpqaDiamond": 49.8,
          "aaHle": 3.6,
          "aaOmniscienceIndex": -54.2,
          "omniscienceAccuracy": 19,
          "omniscienceHallucinationRate": 90.2
        },
        "multilingual": {},
        "instructionFollowing": {
          "aaIfBench": 47.1
        },
        "math": {}
      }
    },
    {
      "slug": "ultravox-v0-7",
      "canonicalModelKey": "ultravox-v0-7",
      "model": "Ultravox v0.7",
      "creator": "Fixie AI",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/ultravox-v0-7",
      "markdownUrl": "https://benchlm.ai/md/models/ultravox-v0-7.md",
      "id": 351,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "ultravox-v0-7",
        "familyName": "Ultravox v0.7",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "ultravox-v0-7",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    },
    {
      "slug": "varco",
      "canonicalModelKey": "varco",
      "model": "Varco",
      "creator": "NC AI",
      "sourceType": "Proprietary",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "64K",
      "contextWindowTokens": 64000,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/varco",
      "markdownUrl": "https://benchlm.ai/md/models/varco.md",
      "id": 213,
      "releaseDate": null,
      "market": "Korea",
      "isRegional": true,
      "family": {
        "familyKey": "varco",
        "familyName": "Varco",
        "variantType": null,
        "snapshotLabel": null,
        "baseFamilyModelKey": "varco",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {},
        "korean": {}
      }
    },
    {
      "slug": "voxtral-4b-tts-2603",
      "canonicalModelKey": "voxtral-4b-tts-2603",
      "model": "Voxtral 4B TTS 2603",
      "creator": "Mistral AI",
      "sourceType": "Open Weight",
      "reasoningType": "Non-Reasoning",
      "contextWindow": "N/A",
      "contextWindowTokens": 0,
      "displayScore": null,
      "provisionalDisplayScore": 0,
      "publicRankingMode": "bench-align-v5",
      "evidenceStatus": null,
      "scoreInterval90": null,
      "rankingEligible": false,
      "overallRank": null,
      "url": "https://benchlm.ai/models/voxtral-4b-tts-2603",
      "markdownUrl": "https://benchlm.ai/md/models/voxtral-4b-tts-2603.md",
      "id": 381,
      "releaseDate": null,
      "market": null,
      "isRegional": false,
      "family": {
        "familyKey": "voxtral-4b-tts-2603",
        "familyName": "Voxtral 4B TTS 2603",
        "variantType": "base",
        "snapshotLabel": null,
        "baseFamilyModelKey": "voxtral-4b-tts-2603",
        "relatedModelKeys": [],
        "isCanonicalFamilyEntry": true,
        "supersedesModelKey": null
      },
      "scores": {
        "displayScore": 0,
        "overallScore": 0,
        "rawOverallScore": 0,
        "verifiedDisplayScore": null,
        "displayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        },
        "verifiedDisplayCategoryScores": {
          "agentic": null,
          "coding": null,
          "reasoning": null,
          "multimodalGrounded": null,
          "knowledge": null,
          "multilingual": null,
          "instructionFollowing": null,
          "math": null
        }
      },
      "ranking": {
        "rankingEligible": false,
        "verifiedRankingEligible": false,
        "overallRank": null,
        "categoryRanks": {},
        "categoryRankingEligible": {
          "agentic": false,
          "coding": false,
          "reasoning": false,
          "multimodalGrounded": false,
          "knowledge": false,
          "multilingual": false,
          "instructionFollowing": false,
          "math": false
        }
      },
      "coverage": {
        "trustedBenchmarkCount": 0,
        "verifiedBenchmarkCount": 0,
        "rankableBenchmarkCount": 0,
        "generatedBenchmarkCount": 0,
        "scoreConfidence": 1
      },
      "benchmarks": {
        "agentic": {},
        "coding": {},
        "reasoning": {},
        "multimodalGrounded": {},
        "knowledge": {},
        "multilingual": {},
        "instructionFollowing": {},
        "math": {}
      }
    }
  ]
}
