{
 "source": "https://ai.paski.dev",
 "generated": "2026-07-27",
 "terms": "Free to use with attribution to ai.paski.dev. Benchmark figures belong to whoever published them — each score carries its own source_url. Prices are vendor list rates and go stale; check the source before relying on one.",
 "counts": {
  "models": 112,
  "benchmarks": 7,
  "published_scores": 278,
  "measured_scores": 88
 },
 "benchmarks": [
  {
   "id": "mmlu-pro",
   "name": "MMLU-Pro",
   "group": "classic",
   "what_it_measures": "Harder MMLU successor: ten answer options per question and a reasoning-heavy question mix.",
   "url": "https://arxiv.org/abs/2406.01574"
  },
  {
   "id": "gpqa",
   "name": "GPQA",
   "group": "classic",
   "what_it_measures": "Graduate-level, Google-proof science questions — the 198-question Diamond subset. Anthropic calls it saturated and is phasing it out.",
   "url": "https://arxiv.org/abs/2311.12022"
  },
  {
   "id": "aime",
   "name": "AIME 2025",
   "group": "classic",
   "what_it_measures": "American Invitational Mathematics Examination 2025, pass@1 with no tools. Saturating: the best models now score near 100.",
   "url": "https://artofproblemsolving.com/wiki/index.php/2025_AIME_I"
  },
  {
   "id": "swe-bench",
   "name": "SWE-bench",
   "group": "agentic",
   "what_it_measures": "Resolving real GitHub issues end-to-end, on the 500-problem human-validated Verified subset.",
   "url": "https://openai.com/index/introducing-swe-bench-verified/"
  },
  {
   "id": "swe-bench-pro",
   "name": "SWE-bench Pro",
   "group": "agentic",
   "what_it_measures": "Harder SWE-bench successor built from longer, multi-file tasks in repositories the models were not trained on. A different benchmark from Verified, never comparable to it.",
   "url": "https://scale.com/research/swe-bench-pro"
  },
  {
   "id": "hle",
   "name": "HLE",
   "group": "agentic",
   "what_it_measures": "Humanity's Last Exam: expert-written questions at the edge of human knowledge. Recorded without tools or search — the augmented figures vendors also publish are excluded.",
   "url": "https://agi.safe.ai/"
  },
  {
   "id": "browsecomp",
   "name": "BrowseComp",
   "group": "agentic",
   "what_it_measures": "Finding hard-to-locate facts on the live web, single agent. Multi-agent and context-managed variants are excluded.",
   "url": "https://openai.com/index/browsecomp/"
  }
 ],
 "models": [
  {
   "id": "claude-3-5-sonnet",
   "name": "Claude 3.5 Sonnet",
   "org": "Anthropic",
   "release_date": "2024-06-20",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://www.anthropic.com/news/claude-3-5-sonnet"
   },
   "notes": "Anthropic's mid-tier model that outperformed the prior flagship Claude 3 Opus at twice the speed and a fifth of the cost.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 59.4,
     "source_url": "https://www-cdn.anthropic.com/fed9cc193a14b84131812372d8d5857f8f304c52/Model_Card_Claude_3_Addendum.pdf"
    }
   ]
  },
  {
   "id": "claude-3-7-sonnet",
   "name": "Claude 3.7 Sonnet",
   "org": "Anthropic",
   "release_date": "2025-02-24",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://www.anthropic.com/news/claude-3-7-sonnet"
   },
   "notes": "Anthropic's first hybrid reasoning model, letting users toggle standard and extended-thinking modes. Anthropic publishes no single-attempt extended-thinking GPQA figure for this model, so the standard-mode score is recorded.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 68,
     "source_url": "https://www.anthropic.com/news/claude-3-7-sonnet"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 63.7,
     "source_url": "https://www.anthropic.com/news/claude-3-7-sonnet"
    }
   ],
   "external": [
    {
     "benchmark_id": "swe-bench",
     "score": 52.8,
     "platform": "swe-bench",
     "variant": "mini-SWE-agent + Claude 3.7 Sonnet (20250219), verified by the maintainers",
     "source_url": "https://www.swebench.com"
    }
   ]
  },
  {
   "id": "claude-fable-5",
   "name": "Claude Fable 5",
   "org": "Anthropic",
   "release_date": "2026-06-09",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://www.anthropic.com/news/claude-fable-5-mythos-5"
   },
   "notes": "Anthropic's most capable generally available model at launch, built from the same checkpoint as Claude Mythos 5 but with broader safety classifiers active.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "swe-bench",
     "score": 95,
     "source_url": "https://www-cdn.anthropic.com/d00db56fa754a1b115b6dd7cb2e3c342ee809620.pdf",
     "effort": "thinking",
     "effort_note": "standard SWE-bench configuration, thinking blocks included in the sampling results"
    },
    {
     "benchmark_id": "swe-bench-pro",
     "score": 80,
     "source_url": "https://www-cdn.anthropic.com/d00db56fa754a1b115b6dd7cb2e3c342ee809620.pdf",
     "effort": "thinking",
     "effort_note": "standard SWE-bench configuration, thinking blocks included in the sampling results"
    }
   ],
   "external": [
    {
     "benchmark_id": "hle",
     "score": 53.3,
     "platform": "artificial-analysis",
     "variant": "Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)",
     "source_url": "https://artificialanalysis.ai/evaluations/humanitys-last-exam"
    }
   ]
  },
  {
   "id": "claude-haiku-4-5",
   "name": "Claude Haiku 4.5",
   "org": "Anthropic",
   "release_date": "2025-10-15",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://www.anthropic.com/news/claude-haiku-4-5"
   },
   "notes": "Anthropic's smallest and fastest model in the Claude 4.5 generation, offering near-Sonnet-4 coding performance at roughly a third of the cost.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 73,
     "source_url": "https://www.anthropic.com/news/claude-haiku-4-5"
    },
    {
     "benchmark_id": "aime",
     "score": 80.7,
     "source_url": "https://www.anthropic.com/news/claude-haiku-4-5"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 73.3,
     "source_url": "https://www.anthropic.com/news/claude-haiku-4-5"
    }
   ],
   "external": [
    {
     "benchmark_id": "swe-bench",
     "score": 66.6,
     "platform": "swe-bench",
     "variant": "mini-SWE-agent + Claude 4.5 Haiku (high reasoning)",
     "source_url": "https://www.swebench.com"
    }
   ]
  },
  {
   "id": "claude-mythos-5",
   "name": "Claude Mythos 5",
   "org": "Anthropic",
   "release_date": "2026-06-09",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://www.anthropic.com/news/claude-fable-5-mythos-5"
   },
   "notes": "Limited-availability counterpart to Claude Fable 5, sharing the same underlying model with some safeguards lifted for approved cyber-defence and life-sciences customers.",
   "lineage": [
    {
     "parent_id": "claude-fable-5",
     "relation": "base"
    }
   ],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 94.1,
     "source_url": "https://www-cdn.anthropic.com/d00db56fa754a1b115b6dd7cb2e3c342ee809620.pdf"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 95.5,
     "source_url": "https://www-cdn.anthropic.com/d00db56fa754a1b115b6dd7cb2e3c342ee809620.pdf",
     "effort": "thinking",
     "effort_note": "standard SWE-bench configuration, thinking blocks included in the sampling results"
    },
    {
     "benchmark_id": "swe-bench-pro",
     "score": 80.3,
     "source_url": "https://www-cdn.anthropic.com/d00db56fa754a1b115b6dd7cb2e3c342ee809620.pdf",
     "effort": "thinking",
     "effort_note": "standard SWE-bench configuration, thinking blocks included in the sampling results"
    },
    {
     "benchmark_id": "hle",
     "score": 59,
     "source_url": "https://www-cdn.anthropic.com/d00db56fa754a1b115b6dd7cb2e3c342ee809620.pdf",
     "effort": "auto",
     "effort_note": "reasoning-only, no tools; thinking set to auto, total tokens capped at 1M"
    },
    {
     "benchmark_id": "browsecomp",
     "score": 88,
     "source_url": "https://www-cdn.anthropic.com/d00db56fa754a1b115b6dd7cb2e3c342ee809620.pdf",
     "effort": "max",
     "effort_note": "adaptive thinking at maximum effort, 10M-token limit, single-agent harness"
    }
   ]
  },
  {
   "id": "claude-opus-4",
   "name": "Claude Opus 4",
   "org": "Anthropic",
   "release_date": "2025-05-22",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://www.anthropic.com/news/claude-4"
   },
   "notes": "Anthropic's flagship model from the Claude 4 generation, positioned at launch as its best coding model. Scores are extended-thinking, single attempt, no tools.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 79.6,
     "source_url": "https://www.anthropic.com/news/claude-4",
     "effort": "thinking",
     "effort_note": "extended thinking, up to 64K tokens"
    },
    {
     "benchmark_id": "aime",
     "score": 75.5,
     "source_url": "https://www.anthropic.com/news/claude-4",
     "effort": "thinking",
     "effort_note": "extended thinking, up to 64K tokens"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 72.5,
     "source_url": "https://www.anthropic.com/news/claude-4",
     "effort": "none",
     "effort_note": "no extended thinking"
    }
   ],
   "external": [
    {
     "benchmark_id": "swe-bench",
     "score": 67.6,
     "platform": "swe-bench",
     "variant": "mini-SWE-agent + Claude 4 Opus (20250514), verified by the maintainers",
     "source_url": "https://www.swebench.com"
    }
   ]
  },
  {
   "id": "claude-opus-4-5",
   "name": "Claude Opus 4.5",
   "org": "Anthropic",
   "release_date": "2025-11-24",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://www.anthropic.com/news/claude-opus-4-5"
   },
   "notes": "Flagship Opus-tier model succeeding Opus 4.1, launched as Anthropic's best model at the time for coding, agents and computer use.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 87,
     "source_url": "https://www.anthropic.com/news/claude-opus-4-5"
    },
    {
     "benchmark_id": "aime",
     "score": 92.77,
     "source_url": "https://assets.anthropic.com/m/64823ba7485345a7/Claude-Opus-4-5-System-Card.pdf"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 80.9,
     "source_url": "https://www.anthropic.com/news/claude-opus-4-5"
    },
    {
     "benchmark_id": "swe-bench-pro",
     "score": 52,
     "source_url": "https://www-cdn.anthropic.com/bf10f64990cfda0ba858290be7b8cc6317685f47.pdf"
    }
   ],
   "external": [
    {
     "benchmark_id": "swe-bench",
     "score": 76.8,
     "platform": "swe-bench",
     "variant": "mini-SWE-agent + Claude 4.5 Opus (high reasoning)",
     "source_url": "https://www.swebench.com"
    }
   ]
  },
  {
   "id": "claude-opus-4-6",
   "name": "Claude Opus 4.6",
   "org": "Anthropic",
   "release_date": "2026-02-05",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://www.anthropic.com/news/claude-opus-4-6"
   },
   "notes": "Opus upgrade introducing coordinated multi-agent teams and deeper document integration, with improved reliability on long coding tasks.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 91.31,
     "source_url": "https://www-cdn.anthropic.com/6a5fa276ac68b9aeb0c8b6af5fa36326e0e166dd.pdf"
    },
    {
     "benchmark_id": "aime",
     "score": 99.79,
     "source_url": "https://www-cdn.anthropic.com/6a5fa276ac68b9aeb0c8b6af5fa36326e0e166dd.pdf"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 80.84,
     "source_url": "https://www-cdn.anthropic.com/6a5fa276ac68b9aeb0c8b6af5fa36326e0e166dd.pdf"
    },
    {
     "benchmark_id": "swe-bench-pro",
     "score": 53.4,
     "source_url": "https://www-cdn.anthropic.com/037f06850df7fbe871e206dad004c3db5fd50340/Claude%20Opus%204.7%20System%20Card.pdf"
    },
    {
     "benchmark_id": "hle",
     "score": 40,
     "source_url": "https://www-cdn.anthropic.com/037f06850df7fbe871e206dad004c3db5fd50340/Claude%20Opus%204.7%20System%20Card.pdf"
    },
    {
     "benchmark_id": "browsecomp",
     "score": 83.7,
     "source_url": "https://www-cdn.anthropic.com/037f06850df7fbe871e206dad004c3db5fd50340/Claude%20Opus%204.7%20System%20Card.pdf"
    }
   ],
   "external": [
    {
     "benchmark_id": "swe-bench",
     "score": 75.6,
     "platform": "swe-bench",
     "variant": "mini-SWE-agent + Claude Opus 4.6",
     "source_url": "https://www.swebench.com"
    }
   ]
  },
  {
   "id": "claude-opus-4-7",
   "name": "Claude Opus 4.7",
   "org": "Anthropic",
   "release_date": "2026-04-16",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://www.anthropic.com/news/claude-opus-4-7"
   },
   "notes": "Opus upgrade emphasising reliability on long-running coding tasks and self-verification, trained with deliberately reduced cyber capability.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 94.2,
     "source_url": "https://www-cdn.anthropic.com/037f06850df7fbe871e206dad004c3db5fd50340/Claude%20Opus%204.7%20System%20Card.pdf"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 87.6,
     "source_url": "https://www-cdn.anthropic.com/037f06850df7fbe871e206dad004c3db5fd50340/Claude%20Opus%204.7%20System%20Card.pdf",
     "effort": "thinking",
     "effort_note": "standard SWE-bench configuration, thinking blocks included in the sampling results; averaged over 5 trials"
    },
    {
     "benchmark_id": "swe-bench-pro",
     "score": 64.3,
     "source_url": "https://www-cdn.anthropic.com/037f06850df7fbe871e206dad004c3db5fd50340/Claude%20Opus%204.7%20System%20Card.pdf",
     "effort": "thinking",
     "effort_note": "standard SWE-bench configuration, thinking blocks included in the sampling results; averaged over 5 trials"
    },
    {
     "benchmark_id": "hle",
     "score": 46.9,
     "source_url": "https://www-cdn.anthropic.com/037f06850df7fbe871e206dad004c3db5fd50340/Claude%20Opus%204.7%20System%20Card.pdf",
     "effort": "max",
     "effort_note": "reasoning-only, no tools; thinking set to auto, total tokens capped at 1M, at max reasoning effort"
    },
    {
     "benchmark_id": "browsecomp",
     "score": 79.3,
     "source_url": "https://www-cdn.anthropic.com/037f06850df7fbe871e206dad004c3db5fd50340/Claude%20Opus%204.7%20System%20Card.pdf",
     "effort": "max",
     "effort_note": "thinking off at max effort, 10M-token limit with context compaction from 200k"
    }
   ]
  },
  {
   "id": "claude-opus-4-8",
   "name": "Claude Opus 4.8",
   "org": "Anthropic",
   "release_date": "2026-05-28",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://www.anthropic.com/news/claude-opus-4-8"
   },
   "notes": "Opus upgrade focused on agentic reliability and honesty, adding a dynamic workflow feature that runs multiple subagents concurrently.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 93.6,
     "source_url": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 88.6,
     "source_url": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf"
    },
    {
     "benchmark_id": "swe-bench-pro",
     "score": 69.2,
     "source_url": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf"
    },
    {
     "benchmark_id": "hle",
     "score": 49.8,
     "source_url": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf"
    },
    {
     "benchmark_id": "browsecomp",
     "score": 84.3,
     "source_url": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf"
    }
   ]
  },
  {
   "id": "claude-opus-5",
   "name": "Claude Opus 5",
   "org": "Anthropic",
   "release_date": "2026-07-24",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://www.anthropic.com/news/claude-opus-5"
   },
   "notes": "Successor to Opus 4.8 adding a low, medium and high effort control, positioned close to Claude Fable 5's frontier intelligence at roughly half the price.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "swe-bench",
     "score": 96,
     "source_url": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf"
    },
    {
     "benchmark_id": "swe-bench-pro",
     "score": 79.2,
     "source_url": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf"
    },
    {
     "benchmark_id": "hle",
     "score": 56.3,
     "source_url": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf"
    },
    {
     "benchmark_id": "browsecomp",
     "score": 90.8,
     "source_url": "https://www-cdn.anthropic.com/c5fbac3f0b1280a933ebd26d3cb8bb9f5bdeaf48/Claude%20Opus%205%20System%20Card.pdf"
    }
   ],
   "external": [
    {
     "benchmark_id": "gpqa",
     "score": 93.7,
     "platform": "artificial-analysis",
     "variant": "Claude Opus 5 (Adaptive Reasoning, High Effort)",
     "source_url": "https://artificialanalysis.ai/evaluations/gpqa-diamond"
    },
    {
     "benchmark_id": "hle",
     "score": 52.6,
     "platform": "artificial-analysis",
     "variant": "Claude Opus 5 (Adaptive Reasoning, Max Effort)",
     "source_url": "https://artificialanalysis.ai/evaluations/humanitys-last-exam"
    }
   ]
  },
  {
   "id": "claude-sonnet-4",
   "name": "Claude Sonnet 4",
   "org": "Anthropic",
   "release_date": "2025-05-22",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://www.anthropic.com/news/claude-4"
   },
   "notes": "Anthropic's mid-tier Claude 4 model, succeeding Claude 3.7 Sonnet with improved coding and instruction-following. Scores are extended-thinking, single attempt, no tools.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 75.4,
     "source_url": "https://www.anthropic.com/news/claude-4",
     "effort": "thinking",
     "effort_note": "extended thinking, up to 64K tokens"
    },
    {
     "benchmark_id": "aime",
     "score": 70.5,
     "source_url": "https://www.anthropic.com/news/claude-4",
     "effort": "thinking",
     "effort_note": "extended thinking, up to 64K tokens"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 72.7,
     "source_url": "https://www.anthropic.com/news/claude-4",
     "effort": "none",
     "effort_note": "no extended thinking"
    }
   ],
   "external": [
    {
     "benchmark_id": "swe-bench",
     "score": 64.9,
     "platform": "swe-bench",
     "variant": "mini-SWE-agent + Claude 4 Sonnet (20250514), verified by the maintainers",
     "source_url": "https://www.swebench.com"
    }
   ]
  },
  {
   "id": "claude-sonnet-4-5",
   "name": "Claude Sonnet 4.5",
   "org": "Anthropic",
   "release_date": "2025-09-29",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://www.anthropic.com/news/claude-sonnet-4-5"
   },
   "notes": "Anthropic's coding-focused model that set a new state of the art on SWE-bench Verified at launch. The 82.0% figure reported with additional test-time compute is deliberately not recorded.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "swe-bench",
     "score": 77.2,
     "source_url": "https://www.anthropic.com/news/claude-sonnet-4-5"
    }
   ],
   "external": [
    {
     "benchmark_id": "swe-bench",
     "score": 71.4,
     "platform": "swe-bench",
     "variant": "mini-SWE-agent + Claude 4.5 Sonnet (high reasoning)",
     "source_url": "https://www.swebench.com"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 70.6,
     "platform": "swe-bench",
     "variant": "mini-SWE-agent + Claude 4.5 Sonnet (20250929), verified by the maintainers",
     "source_url": "https://www.swebench.com"
    }
   ]
  },
  {
   "id": "claude-sonnet-4-6",
   "name": "Claude Sonnet 4.6",
   "org": "Anthropic",
   "release_date": "2026-02-17",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://www.anthropic.com/news/claude-sonnet-4-6"
   },
   "notes": "Mid-tier Sonnet upgrade with a 1M-token beta context window, approaching Opus-level intelligence at Sonnet 4.5 pricing.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 89.9,
     "source_url": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf"
    },
    {
     "benchmark_id": "aime",
     "score": 95.6,
     "source_url": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 79.6,
     "source_url": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf"
    },
    {
     "benchmark_id": "swe-bench-pro",
     "score": 58.1,
     "source_url": "https://www-cdn.anthropic.com/480e0bb54327b9622282e9c39a83a4f490ed377e/Claude%20Sonnet%205%20System%20Card.pdf"
    },
    {
     "benchmark_id": "hle",
     "score": 34.6,
     "source_url": "https://www-cdn.anthropic.com/480e0bb54327b9622282e9c39a83a4f490ed377e/Claude%20Sonnet%205%20System%20Card.pdf"
    },
    {
     "benchmark_id": "browsecomp",
     "score": 74,
     "source_url": "https://www-cdn.anthropic.com/bbd8ef16d70b7a1665f14f306ee88b53f686aa75/Claude%20Sonnet%204.6%20System%20Card.pdf"
    }
   ]
  },
  {
   "id": "claude-sonnet-5",
   "name": "Claude Sonnet 5",
   "org": "Anthropic",
   "release_date": "2026-06-30",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://www.anthropic.com/news/claude-sonnet-5"
   },
   "notes": "Most agentic Sonnet-tier model to date, positioned as a lower-cost option approaching Opus-class performance on many agentic tasks.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "swe-bench",
     "score": 85.2,
     "source_url": "https://www-cdn.anthropic.com/480e0bb54327b9622282e9c39a83a4f490ed377e/Claude%20Sonnet%205%20System%20Card.pdf",
     "effort": "thinking",
     "effort_note": "standard SWE-bench configuration, thinking blocks included in the sampling results"
    },
    {
     "benchmark_id": "swe-bench-pro",
     "score": 63.2,
     "source_url": "https://www-cdn.anthropic.com/480e0bb54327b9622282e9c39a83a4f490ed377e/Claude%20Sonnet%205%20System%20Card.pdf",
     "effort": "thinking",
     "effort_note": "standard SWE-bench configuration, thinking blocks included in the sampling results"
    },
    {
     "benchmark_id": "hle",
     "score": 43.2,
     "source_url": "https://www-cdn.anthropic.com/480e0bb54327b9622282e9c39a83a4f490ed377e/Claude%20Sonnet%205%20System%20Card.pdf",
     "effort": "auto",
     "effort_note": "reasoning-only, no tools; thinking set to auto, total tokens capped at 1M"
    },
    {
     "benchmark_id": "browsecomp",
     "score": 84.7,
     "source_url": "https://www-cdn.anthropic.com/480e0bb54327b9622282e9c39a83a4f490ed377e/Claude%20Sonnet%205%20System%20Card.pdf",
     "effort": "max",
     "effort_note": "adaptive thinking at maximum effort, 10M-token limit with context compaction from 200k"
    }
   ]
  },
  {
   "hf_id": "deepseek-ai/DeepSeek-R1",
   "id": "deepseek-r1",
   "kind": "llm",
   "license": "mit",
   "lineage": [],
   "links": {
    "hf": "https://huggingface.co/deepseek-ai/DeepSeek-R1"
   },
   "name": "DeepSeek R1",
   "notes": "Open-weights reasoning model.",
   "org": "DeepSeek",
   "params": "671B",
   "release_date": "2025-01-20",
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 71.5,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-R1"
    }
   ]
  },
  {
   "id": "deepseek-r1-0528",
   "name": "DeepSeek R1-0528",
   "org": "DeepSeek",
   "release_date": "2025-05-28",
   "params": "671B-A37B",
   "license": "mit",
   "kind": "llm",
   "hf_id": "deepseek-ai/DeepSeek-R1-0528",
   "links": {
    "hf": "https://huggingface.co/deepseek-ai/DeepSeek-R1-0528"
   },
   "notes": "A May 2025 checkpoint update to DeepSeek R1 with substantially deeper chain-of-thought reasoning and large gains on math, coding, and general-knowledge benchmarks.",
   "lineage": [
    {
     "parent_id": "deepseek-r1",
     "relation": "finetune"
    }
   ],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 85,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-R1-0528"
    },
    {
     "benchmark_id": "gpqa",
     "score": 81,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-R1-0528"
    },
    {
     "benchmark_id": "aime",
     "score": 87.5,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-R1-0528"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 57.6,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-R1-0528"
    }
   ]
  },
  {
   "hf_id": "deepseek-ai/DeepSeek-R1-Distill-Llama-8B",
   "id": "deepseek-r1-distill-llama-8b",
   "kind": "llm",
   "license": "mit",
   "lineage": [
    {
     "parent_id": "deepseek-r1",
     "relation": "distill"
    },
    {
     "parent_id": "llama-3-1-8b",
     "relation": "base"
    }
   ],
   "links": {
    "hf": "https://huggingface.co/deepseek-ai/DeepSeek-R1-Distill-Llama-8B"
   },
   "name": "DeepSeek R1 Distill Llama 8B",
   "notes": "R1 reasoning distilled into Llama 3.1 8B.",
   "org": "DeepSeek",
   "params": "8B",
   "release_date": "2025-01-20",
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 49,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-R1-Distill-Llama-8B"
    }
   ]
  },
  {
   "id": "deepseek-r1-distill-qwen-32b",
   "name": "DeepSeek R1-Distill-Qwen-32B",
   "org": "DeepSeek",
   "release_date": "2025-01-20",
   "params": "32B",
   "license": "mit",
   "kind": "llm",
   "hf_id": "deepseek-ai/DeepSeek-R1-Distill-Qwen-32B",
   "links": {
    "hf": "https://huggingface.co/deepseek-ai/DeepSeek-R1-Distill-Qwen-32B",
    "paper": "https://arxiv.org/abs/2501.12948"
   },
   "notes": "A dense 32B reasoning model produced by distilling DeepSeek R1's reasoning traces onto a Qwen2.5-32B base, released alongside R1 itself.",
   "lineage": [
    {
     "parent_id": "deepseek-r1",
     "relation": "distill"
    }
   ],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 62.1,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-R1-Distill-Qwen-32B"
    }
   ]
  },
  {
   "id": "deepseek-v3",
   "name": "DeepSeek V3",
   "org": "DeepSeek",
   "release_date": "2024-12-26",
   "params": "671B-A37B",
   "license": "mit",
   "kind": "llm",
   "hf_id": "deepseek-ai/DeepSeek-V3",
   "links": {
    "hf": "https://huggingface.co/deepseek-ai/DeepSeek-V3",
    "paper": "https://arxiv.org/abs/2412.19437"
   },
   "notes": "A 671B-parameter mixture-of-experts model trained with multi-token prediction and FP8 precision, DeepSeek's flagship non-reasoning chat model released in December 2024.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 75.9,
     "source_url": "https://github.com/deepseek-ai/DeepSeek-V3"
    },
    {
     "benchmark_id": "gpqa",
     "score": 59.1,
     "source_url": "https://github.com/deepseek-ai/DeepSeek-V3"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 42,
     "source_url": "https://github.com/deepseek-ai/DeepSeek-V3"
    }
   ]
  },
  {
   "id": "deepseek-v3-1",
   "name": "DeepSeek V3.1",
   "org": "DeepSeek",
   "release_date": "2025-08-21",
   "params": "671B-A37B",
   "license": "mit",
   "kind": "llm",
   "hf_id": "deepseek-ai/DeepSeek-V3.1",
   "links": {
    "hf": "https://huggingface.co/deepseek-ai/DeepSeek-V3.1",
    "blog": "https://api-docs.deepseek.com/news/news250821"
   },
   "notes": "A hybrid model merging DeepSeek V3 and R1 into a single checkpoint with a switchable thinking and non-thinking chat template.",
   "lineage": [
    {
     "parent_id": "deepseek-v3",
     "relation": "finetune"
    }
   ],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 80.1,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V3.1"
    },
    {
     "benchmark_id": "aime",
     "score": 88.4,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V3.1"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 66,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V3.1"
    },
    {
     "benchmark_id": "hle",
     "score": 15.9,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V3.1"
    },
    {
     "benchmark_id": "browsecomp",
     "score": 30,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V3.1"
    }
   ]
  },
  {
   "id": "deepseek-v3-2",
   "name": "DeepSeek V3.2",
   "org": "DeepSeek",
   "release_date": "2025-12-01",
   "params": "685B",
   "license": "mit",
   "kind": "llm",
   "hf_id": "deepseek-ai/DeepSeek-V3.2",
   "links": {
    "hf": "https://huggingface.co/deepseek-ai/DeepSeek-V3.2"
   },
   "notes": "Stable production successor to V3.2-Exp, combining DeepSeek Sparse Attention with a scaled reinforcement-learning framework and agentic tool-use training.",
   "lineage": [
    {
     "parent_id": "deepseek-v3-2-exp",
     "relation": "finetune"
    }
   ],
   "scores": [
    {
     "benchmark_id": "swe-bench",
     "score": 73.1,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V3.2"
    },
    {
     "benchmark_id": "hle",
     "score": 25.1,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V3.2"
    }
   ],
   "external": [
    {
     "benchmark_id": "swe-bench",
     "score": 70,
     "platform": "swe-bench",
     "variant": "mini-SWE-agent + DeepSeek V3.2 (high reasoning)",
     "source_url": "https://www.swebench.com"
    }
   ]
  },
  {
   "id": "deepseek-v3-2-exp",
   "name": "DeepSeek V3.2-Exp",
   "org": "DeepSeek",
   "release_date": "2025-09-29",
   "params": "671B-A37B",
   "license": "mit",
   "kind": "llm",
   "hf_id": "deepseek-ai/DeepSeek-V3.2-Exp",
   "links": {
    "hf": "https://huggingface.co/deepseek-ai/DeepSeek-V3.2-Exp",
    "blog": "https://api-docs.deepseek.com/news/news250929"
   },
   "notes": "An experimental update to V3.1-Terminus introducing DeepSeek Sparse Attention for more efficient long-context training and inference at roughly the same output quality.",
   "lineage": [
    {
     "parent_id": "deepseek-v3-1",
     "relation": "finetune"
    }
   ],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 85,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V3.2-Exp"
    },
    {
     "benchmark_id": "gpqa",
     "score": 79.9,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V3.2-Exp"
    },
    {
     "benchmark_id": "aime",
     "score": 89.3,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V3.2-Exp"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 67.8,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V3.2-Exp"
    },
    {
     "benchmark_id": "hle",
     "score": 19.8,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V3.2-Exp"
    },
    {
     "benchmark_id": "browsecomp",
     "score": 40.1,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V3.2-Exp"
    }
   ]
  },
  {
   "id": "deepseek-v3-2-speciale",
   "name": "DeepSeek V3.2-Speciale",
   "org": "DeepSeek",
   "release_date": "2025-11-28",
   "params": "685B",
   "license": "mit",
   "kind": "llm",
   "hf_id": "deepseek-ai/DeepSeek-V3.2-Speciale",
   "links": {
    "hf": "https://huggingface.co/deepseek-ai/DeepSeek-V3.2-Speciale"
   },
   "notes": "Deep-reasoning variant of V3.2 that drops tool calling in favour of maximum reasoning depth, targeting maths and competitive programming.",
   "lineage": [
    {
     "parent_id": "deepseek-v3-2-exp",
     "relation": "finetune"
    }
   ],
   "scores": [
    {
     "benchmark_id": "hle",
     "score": 30.6,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V3.2-Speciale"
    }
   ]
  },
  {
   "id": "deepseek-v4-flash",
   "name": "DeepSeek V4-Flash",
   "org": "DeepSeek",
   "release_date": "2026-04-22",
   "params": "284B-A13B",
   "license": "mit",
   "kind": "llm",
   "hf_id": "deepseek-ai/DeepSeek-V4-Flash",
   "links": {
    "hf": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash"
   },
   "notes": "Smaller, faster sibling of V4-Pro sharing the same 1M-token architecture. Scores are from its Think Max reasoning mode.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 86.2,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash"
    },
    {
     "benchmark_id": "gpqa",
     "score": 88.1,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 79,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash"
    },
    {
     "benchmark_id": "swe-bench-pro",
     "score": 52.6,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash"
    },
    {
     "benchmark_id": "hle",
     "score": 34.8,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash"
    },
    {
     "benchmark_id": "browsecomp",
     "score": 73.2,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash"
    }
   ]
  },
  {
   "id": "deepseek-v4-pro",
   "name": "DeepSeek V4-Pro",
   "org": "DeepSeek",
   "release_date": "2026-04-22",
   "params": "1600B-A49B",
   "license": "mit",
   "kind": "llm",
   "hf_id": "deepseek-ai/DeepSeek-V4-Pro",
   "links": {
    "hf": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro"
   },
   "notes": "DeepSeek's V4 flagship with a 1M-token context, combining Compressed Sparse Attention and Heavily Compressed Attention. Scores are from its Think Max extended-reasoning mode.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 87.5,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro"
    },
    {
     "benchmark_id": "gpqa",
     "score": 90.1,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 80.6,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro"
    },
    {
     "benchmark_id": "swe-bench-pro",
     "score": 55.4,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro"
    },
    {
     "benchmark_id": "hle",
     "score": 37.7,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro"
    },
    {
     "benchmark_id": "browsecomp",
     "score": 83.4,
     "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro"
    }
   ]
  },
  {
   "id": "devstral-2",
   "name": "Devstral 2",
   "org": "Mistral AI",
   "release_date": "2025-12-09",
   "params": "123B",
   "license": "modified-mit",
   "kind": "llm",
   "hf_id": "mistralai/Devstral-2-123B-Instruct-2512",
   "links": {
    "hf": "https://huggingface.co/mistralai/Devstral-2-123B-Instruct-2512",
    "blog": "https://mistral.ai/news/devstral-2-vibe-cli/"
   },
   "notes": "Mistral's flagship open-weight agentic coding model, released alongside the Mistral Vibe CLI.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "swe-bench",
     "score": 72.2,
     "source_url": "https://huggingface.co/mistralai/Devstral-2-123B-Instruct-2512"
    }
   ]
  },
  {
   "id": "devstral-small",
   "name": "Devstral Small 1.1",
   "org": "Mistral AI",
   "release_date": "2025-07-10",
   "params": "24B",
   "license": "apache-2.0",
   "kind": "llm",
   "hf_id": "mistralai/Devstral-Small-2507",
   "links": {
    "hf": "https://huggingface.co/mistralai/Devstral-Small-2507",
    "blog": "https://mistral.ai/news/devstral-2507/"
   },
   "notes": "Open-weight agentic coding model built on Mistral Small 3.1, which Mistral said set a new state of the art for open models on SWE-bench Verified at release.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "swe-bench",
     "score": 53.6,
     "source_url": "https://mistral.ai/news/devstral-2507/"
    }
   ]
  },
  {
   "id": "devstral-small-2",
   "name": "Devstral Small 2",
   "org": "Mistral AI",
   "release_date": "2025-12-09",
   "params": "24B",
   "license": "apache-2.0",
   "kind": "llm",
   "hf_id": "mistralai/Devstral-Small-2-24B-Instruct-2512",
   "links": {
    "hf": "https://huggingface.co/mistralai/Devstral-Small-2-24B-Instruct-2512",
    "blog": "https://mistral.ai/news/devstral-2-vibe-cli/"
   },
   "notes": "Compact open-weight coding and agentic model small enough to run offline on a single laptop.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "swe-bench",
     "score": 68,
     "source_url": "https://huggingface.co/mistralai/Devstral-Small-2-24B-Instruct-2512"
    }
   ]
  },
  {
   "id": "gemini-1-5-pro",
   "name": "Gemini 1.5 Pro",
   "org": "Google",
   "release_date": "2024-05-14",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "paper": "https://arxiv.org/abs/2403.05530",
    "blog": "https://blog.google/products/gemini/google-gemini-update-may-2024/"
   },
   "notes": "Google's sparse mixture-of-experts multimodal model with a long-context window of up to millions of tokens.",
   "lineage": [],
   "scores": [],
   "external": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 65.7,
     "platform": "artificial-analysis",
     "variant": "Gemini 1.5 Pro (May '24)",
     "source_url": "https://artificialanalysis.ai/models/gemini-1-5-pro-may-2024"
    },
    {
     "benchmark_id": "gpqa",
     "score": 37.1,
     "platform": "artificial-analysis",
     "variant": "Gemini 1.5 Pro (May '24)",
     "source_url": "https://artificialanalysis.ai/models/gemini-1-5-pro-may-2024"
    },
    {
     "benchmark_id": "hle",
     "score": 3.9,
     "platform": "artificial-analysis",
     "variant": "Gemini 1.5 Pro (May '24)",
     "source_url": "https://artificialanalysis.ai/models/gemini-1-5-pro-may-2024"
    }
   ]
  },
  {
   "id": "gemini-2-0-flash",
   "name": "Gemini 2.0 Flash",
   "org": "Google",
   "release_date": "2025-02-05",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://developers.googleblog.com/en/gemini-2-family-expands/"
   },
   "notes": "Google's fast multimodal workhorse model with native tool use, the GA successor to Gemini 1.5 Flash.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 77.6,
     "source_url": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-2-0-Flash-Model-Card.pdf"
    },
    {
     "benchmark_id": "gpqa",
     "score": 60.1,
     "source_url": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-2-0-Flash-Model-Card.pdf"
    }
   ],
   "external": [
    {
     "benchmark_id": "swe-bench",
     "score": 13.5,
     "platform": "swe-bench",
     "variant": "mini-SWE-agent + Gemini 2.0 flash, verified by the maintainers",
     "source_url": "https://www.swebench.com"
    }
   ]
  },
  {
   "id": "gemini-2-5-flash",
   "name": "Gemini 2.5 Flash",
   "org": "Google",
   "release_date": "2025-06-17",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://cloud.google.com/blog/products/ai-machine-learning/expanding-gemini-2-5-flash-and-pro-capabilities"
   },
   "notes": "Google's smaller, faster reasoning model, balancing cost and thinking capability within the 2.5 family.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 80.8,
     "source_url": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-2-5-Flash-Model-Card.pdf"
    },
    {
     "benchmark_id": "aime",
     "score": 75.6,
     "source_url": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-2-5-Flash-Model-Card.pdf"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 54,
     "source_url": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-2-5-Flash-Model-Card.pdf"
    },
    {
     "benchmark_id": "hle",
     "score": 11,
     "source_url": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-2-5-Flash-Model-Card.pdf"
    }
   ],
   "external": [
    {
     "benchmark_id": "swe-bench",
     "score": 28.7,
     "platform": "swe-bench",
     "variant": "mini-SWE-agent + Gemini 2.5 Flash (2025-04-17), verified by the maintainers",
     "source_url": "https://www.swebench.com"
    }
   ]
  },
  {
   "id": "gemini-2-5-pro",
   "name": "Gemini 2.5 Pro",
   "org": "Google",
   "release_date": "2025-06-17",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://blog.google/products/gemini/gemini-2-5-pro-updates/"
   },
   "notes": "Google's frontier reasoning model with adaptive thinking budgets and long context.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 86.4,
     "source_url": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-2-5-Pro-Model-Card.pdf"
    },
    {
     "benchmark_id": "aime",
     "score": 88,
     "source_url": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-2-5-Pro-Model-Card.pdf"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 59.6,
     "source_url": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-2-5-Pro-Model-Card.pdf"
    },
    {
     "benchmark_id": "hle",
     "score": 21.6,
     "source_url": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-2-5-Pro-Model-Card.pdf"
    }
   ],
   "external": [
    {
     "benchmark_id": "swe-bench",
     "score": 53.6,
     "platform": "swe-bench",
     "variant": "mini-SWE-agent + Gemini 2.5 Pro (2025-05-06), verified by the maintainers",
     "source_url": "https://www.swebench.com"
    }
   ]
  },
  {
   "id": "gemini-3-1-flash-lite",
   "name": "Gemini 3.1 Flash-Lite",
   "org": "Google",
   "release_date": "2026-03-03",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://deepmind.google/models/model-cards/gemini-3-1-flash-lite/"
   },
   "notes": "Google's most cost-effective Gemini 3 model, stated to be based on Gemini 3 Pro and optimised for high-volume, latency-sensitive tasks.",
   "lineage": [
    {
     "parent_id": "gemini-3-pro",
     "relation": "distill"
    }
   ],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 86.9,
     "source_url": "https://deepmind.google/models/model-cards/gemini-3-1-flash-lite/"
    },
    {
     "benchmark_id": "swe-bench-pro",
     "score": 38.3,
     "source_url": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-5-Flash-Lite-Model-Card.pdf"
    },
    {
     "benchmark_id": "hle",
     "score": 16,
     "source_url": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-1-Flash-Lite-Model-Card.pdf"
    }
   ]
  },
  {
   "id": "gemini-3-1-pro",
   "name": "Gemini 3.1 Pro",
   "org": "Google",
   "release_date": "2026-02-19",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://deepmind.google/models/model-cards/gemini-3-1-pro/"
   },
   "notes": "Next iteration of the Gemini 3 series, stated by Google to be based on Gemini 3 Pro, with a 1M-token multimodal context and stronger agentic coding.",
   "lineage": [
    {
     "parent_id": "gemini-3-pro",
     "relation": "finetune"
    }
   ],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 94.3,
     "source_url": "https://deepmind.google/models/model-cards/gemini-3-1-pro/"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 80.6,
     "source_url": "https://deepmind.google/models/model-cards/gemini-3-1-pro/"
    },
    {
     "benchmark_id": "swe-bench-pro",
     "score": 54.2,
     "source_url": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-1-Pro-Model-Card.pdf"
    },
    {
     "benchmark_id": "hle",
     "score": 44.4,
     "source_url": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-1-Pro-Model-Card.pdf"
    },
    {
     "benchmark_id": "browsecomp",
     "score": 85.9,
     "source_url": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-1-Pro-Model-Card.pdf"
    }
   ],
   "external": [
    {
     "benchmark_id": "gpqa",
     "score": 94.1,
     "platform": "artificial-analysis",
     "variant": "Gemini 3.1 Pro Preview",
     "source_url": "https://artificialanalysis.ai/evaluations/gpqa-diamond"
    }
   ]
  },
  {
   "id": "gemini-3-5-flash",
   "name": "Gemini 3.5 Flash",
   "org": "Google",
   "release_date": "2026-05-19",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-5/"
   },
   "notes": "Google's strongest agentic and coding model at launch, marketed as Pro-level reasoning at Flash-class latency.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "swe-bench-pro",
     "score": 53.9,
     "source_url": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-5-Flash-Model-Card.pdf"
    },
    {
     "benchmark_id": "hle",
     "score": 40.2,
     "source_url": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-5-Flash-Model-Card.pdf"
    }
   ]
  },
  {
   "id": "gemini-3-5-flash-lite",
   "name": "Gemini 3.5 Flash-Lite",
   "org": "Google",
   "release_date": "2026-07-21",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://deepmind.google/models/model-cards/gemini-3-5-flash-lite/"
   },
   "notes": "Fastest model in the Gemini 3.5 series, stated by Google to be based on Gemini 3.1 Flash-Lite.",
   "lineage": [
    {
     "parent_id": "gemini-3-1-flash-lite",
     "relation": "finetune"
    }
   ],
   "scores": [
    {
     "benchmark_id": "swe-bench-pro",
     "score": 54.2,
     "source_url": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-5-Flash-Lite-Model-Card.pdf"
    }
   ]
  },
  {
   "id": "gemini-3-6-flash",
   "name": "Gemini 3.6 Flash",
   "org": "Google",
   "release_date": "2026-07-21",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://deepmind.google/models/model-cards/gemini-3-6-flash/"
   },
   "notes": "Successor to Gemini 3.5 Flash and Google's new default workhorse model, stated to be based on Gemini 3.5 Flash with lower output cost.",
   "lineage": [
    {
     "parent_id": "gemini-3-5-flash",
     "relation": "finetune"
    }
   ],
   "scores": [
    {
     "benchmark_id": "swe-bench-pro",
     "score": 58.7,
     "source_url": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-6-Flash-Model-Card.pdf"
    }
   ]
  },
  {
   "id": "gemini-3-deep-think",
   "name": "Gemini 3 Deep Think",
   "org": "Google",
   "release_date": "2025-12-04",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://blog.google/products/gemini/gemini-3-deep-think/"
   },
   "notes": "Extended-reasoning product built on Gemini 3 Pro for Google AI Ultra subscribers, upgraded in February 2026 with olympiad-level results.",
   "lineage": [
    {
     "parent_id": "gemini-3-pro",
     "relation": "finetune"
    }
   ],
   "scores": []
  },
  {
   "id": "gemini-3-flash",
   "name": "Gemini 3 Flash",
   "org": "Google",
   "release_date": "2025-12-17",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://blog.google/products/gemini/gemini-3-flash/"
   },
   "notes": "Cost-efficient, low-latency member of the Gemini 3 family combining Pro-grade reasoning with Flash-level speed.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 90.4,
     "source_url": "https://blog.google/products/gemini/gemini-3-flash/"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 78,
     "source_url": "https://blog.google/products/gemini/gemini-3-flash/"
    },
    {
     "benchmark_id": "swe-bench-pro",
     "score": 48.4,
     "source_url": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-5-Flash-Model-Card.pdf"
    },
    {
     "benchmark_id": "hle",
     "score": 33.7,
     "source_url": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-Flash-Model-Card.pdf"
    }
   ],
   "external": [
    {
     "benchmark_id": "swe-bench",
     "score": 75.8,
     "platform": "swe-bench",
     "variant": "mini-SWE-agent + Gemini 3 Flash (high reasoning)",
     "source_url": "https://www.swebench.com"
    }
   ]
  },
  {
   "id": "gemini-3-pro",
   "name": "Gemini 3 Pro",
   "org": "Google",
   "release_date": "2025-11-18",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://blog.google/products-and-platforms/products/gemini/gemini-3/"
   },
   "notes": "Google's next-generation frontier multimodal reasoning model, succeeding Gemini 2.5 Pro.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 91.9,
     "source_url": "https://blog.google/products-and-platforms/products/gemini/gemini-3/"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 76.2,
     "source_url": "https://blog.google/products-and-platforms/products/gemini/gemini-3/"
    },
    {
     "benchmark_id": "swe-bench-pro",
     "score": 43.3,
     "source_url": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-1-Pro-Model-Card.pdf"
    },
    {
     "benchmark_id": "hle",
     "score": 37.5,
     "source_url": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-Pro-Model-Card.pdf"
    },
    {
     "benchmark_id": "browsecomp",
     "score": 59.2,
     "source_url": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-1-Pro-Model-Card.pdf"
    }
   ],
   "external": [
    {
     "benchmark_id": "swe-bench",
     "score": 69.6,
     "platform": "swe-bench",
     "variant": "mini-SWE-agent + Gemini 3 Pro",
     "source_url": "https://www.swebench.com"
    }
   ]
  },
  {
   "id": "gemma-2-27b",
   "name": "Gemma 2 27B",
   "org": "Google",
   "release_date": "2024-06-27",
   "params": "27B",
   "license": "gemma",
   "kind": "llm",
   "hf_id": "google/gemma-2-27b",
   "links": {
    "hf": "https://huggingface.co/google/gemma-2-27b",
    "paper": "https://arxiv.org/abs/2408.00118"
   },
   "notes": "Google's open-weights pretrained model trained with knowledge distillation from larger models. Scores are for the pretrained checkpoint.",
   "lineage": [],
   "scores": []
  },
  {
   "id": "gemma-3-27b-it",
   "name": "Gemma 3 27B Instruct",
   "org": "Google",
   "release_date": "2025-03-12",
   "params": "27B",
   "license": "gemma",
   "kind": "llm",
   "hf_id": "google/gemma-3-27b-it",
   "links": {
    "hf": "https://huggingface.co/google/gemma-3-27b-it",
    "paper": "https://arxiv.org/abs/2503.19786"
   },
   "notes": "Google's instruction-tuned open-weights multimodal model, the largest single-accelerator-deployable model in the Gemma 3 family.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 67.5,
     "source_url": "https://arxiv.org/html/2503.19786v1"
    },
    {
     "benchmark_id": "gpqa",
     "score": 42.4,
     "source_url": "https://arxiv.org/html/2503.19786v1"
    }
   ]
  },
  {
   "id": "gemma-4-12b-it",
   "name": "Gemma 4 12B Instruct",
   "org": "Google",
   "release_date": "2026-06-03",
   "params": "12B",
   "license": "apache-2.0",
   "kind": "llm",
   "hf_id": "google/gemma-4-12B-it",
   "links": {
    "hf": "https://huggingface.co/google/gemma-4-12B-it",
    "paper": "https://arxiv.org/abs/2607.02770"
   },
   "notes": "Mid-size Gemma 4 model with an encoder-free unified decoder that ingests raw image and audio patches directly, aimed at local agentic use.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 77.2,
     "source_url": "https://huggingface.co/google/gemma-4-12B-it"
    },
    {
     "benchmark_id": "gpqa",
     "score": 78.8,
     "source_url": "https://huggingface.co/google/gemma-4-12B-it"
    }
   ]
  },
  {
   "id": "gemma-4-26b-a4b-it",
   "name": "Gemma 4 26B A4B Instruct",
   "org": "Google",
   "release_date": "2026-04-02",
   "params": "26B-A4B",
   "license": "apache-2.0",
   "kind": "llm",
   "hf_id": "google/gemma-4-26B-A4B-it",
   "links": {
    "hf": "https://huggingface.co/google/gemma-4-26B-A4B-it",
    "paper": "https://arxiv.org/abs/2607.02770"
   },
   "notes": "Sparse mixture-of-experts variant of Gemma 4 with 128 routed experts, built for efficient multimodal agentic workloads.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 82.6,
     "source_url": "https://huggingface.co/google/gemma-4-26B-A4B-it"
    },
    {
     "benchmark_id": "gpqa",
     "score": 82.3,
     "source_url": "https://huggingface.co/google/gemma-4-26B-A4B-it"
    }
   ]
  },
  {
   "id": "gemma-4-31b-it",
   "name": "Gemma 4 31B Instruct",
   "org": "Google",
   "release_date": "2026-04-02",
   "params": "31B",
   "license": "apache-2.0",
   "kind": "llm",
   "hf_id": "google/gemma-4-31B-it",
   "links": {
    "hf": "https://huggingface.co/google/gemma-4-31B-it",
    "paper": "https://arxiv.org/abs/2607.02770"
   },
   "notes": "Google's flagship dense Gemma 4 model, natively multimodal with a 256K context, released under Apache 2.0 instead of the prior custom Gemma licence.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 85.2,
     "source_url": "https://huggingface.co/google/gemma-4-31B-it"
    },
    {
     "benchmark_id": "gpqa",
     "score": 84.3,
     "source_url": "https://huggingface.co/google/gemma-4-31B-it"
    }
   ]
  },
  {
   "id": "glm-4-5",
   "name": "GLM-4.5",
   "org": "Z.ai",
   "release_date": "2025-07-28",
   "params": "355B-A32B",
   "license": "mit",
   "kind": "llm",
   "hf_id": "zai-org/GLM-4.5",
   "links": {
    "hf": "https://huggingface.co/zai-org/GLM-4.5",
    "paper": "https://arxiv.org/abs/2508.06471"
   },
   "notes": "Z.ai's 355B-parameter (32B active) flagship model unifying agentic, reasoning, and coding capabilities in one checkpoint.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 84.6,
     "source_url": "https://arxiv.org/abs/2508.06471"
    },
    {
     "benchmark_id": "gpqa",
     "score": 79.1,
     "source_url": "https://arxiv.org/abs/2508.06471"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 64.2,
     "source_url": "https://arxiv.org/abs/2508.06471"
    }
   ],
   "external": [
    {
     "benchmark_id": "swe-bench",
     "score": 54.2,
     "platform": "swe-bench",
     "variant": "mini-SWE-agent + GLM-4.5 (2025-08-22), verified by the maintainers",
     "source_url": "https://www.swebench.com"
    }
   ]
  },
  {
   "id": "glm-4-6",
   "name": "GLM-4.6",
   "org": "Z.ai",
   "release_date": "2025-09-30",
   "params": "357B-A32B",
   "license": "mit",
   "kind": "llm",
   "hf_id": "zai-org/GLM-4.6",
   "links": {
    "hf": "https://huggingface.co/zai-org/GLM-4.6"
   },
   "notes": "Z.ai's successor to GLM-4.5, expanding context to 200K tokens and improving coding and agentic performance. Z.ai publishes its benchmark results only as chart images, so no sourced scores are recorded here yet.",
   "lineage": [
    {
     "parent_id": "glm-4-5",
     "relation": "finetune"
    }
   ],
   "scores": [
    {
     "benchmark_id": "hle",
     "score": 17.2,
     "source_url": "https://huggingface.co/zai-org/GLM-4.7"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 68,
     "source_url": "https://huggingface.co/zai-org/GLM-4.7"
    },
    {
     "benchmark_id": "browsecomp",
     "score": 45.1,
     "source_url": "https://huggingface.co/zai-org/GLM-4.7"
    }
   ],
   "external": [
    {
     "benchmark_id": "swe-bench",
     "score": 55.4,
     "platform": "swe-bench",
     "variant": "mini-SWE-agent + GLM-4.6 (T=1), verified by the maintainers",
     "source_url": "https://www.swebench.com"
    }
   ]
  },
  {
   "id": "glm-4-7",
   "name": "GLM-4.7",
   "org": "Z.ai",
   "release_date": "2025-12-22",
   "params": "358B",
   "license": "mit",
   "kind": "llm",
   "hf_id": "zai-org/GLM-4.7",
   "links": {
    "hf": "https://huggingface.co/zai-org/GLM-4.7"
   },
   "notes": "Z.ai's coding-focused flagship with interleaved thinking modes, positioned as the top open-weight coder at release.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 84.3,
     "source_url": "https://huggingface.co/zai-org/GLM-4.7"
    },
    {
     "benchmark_id": "gpqa",
     "score": 85.7,
     "source_url": "https://huggingface.co/zai-org/GLM-4.7"
    },
    {
     "benchmark_id": "aime",
     "score": 95.7,
     "source_url": "https://huggingface.co/zai-org/GLM-4.7"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 73.8,
     "source_url": "https://huggingface.co/zai-org/GLM-4.7"
    },
    {
     "benchmark_id": "hle",
     "score": 24.8,
     "source_url": "https://huggingface.co/zai-org/GLM-4.7"
    },
    {
     "benchmark_id": "browsecomp",
     "score": 52,
     "source_url": "https://huggingface.co/zai-org/GLM-4.7"
    }
   ]
  },
  {
   "id": "glm-4-7-flash",
   "name": "GLM-4.7-Flash",
   "org": "Z.ai",
   "release_date": "2026-01-19",
   "params": "30B-A3B",
   "license": "mit",
   "kind": "llm",
   "hf_id": "zai-org/GLM-4.7-Flash",
   "links": {
    "hf": "https://huggingface.co/zai-org/GLM-4.7-Flash"
   },
   "notes": "Lightweight 30B-A3B mixture-of-experts version of GLM-4.7, offered as Z.ai's free-tier coding and reasoning model.",
   "lineage": [
    {
     "parent_id": "glm-4-7",
     "relation": "distill"
    }
   ],
   "scores": [
    {
     "benchmark_id": "aime",
     "score": 91.6,
     "source_url": "https://huggingface.co/zai-org/GLM-4.7-Flash"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 59.2,
     "source_url": "https://huggingface.co/zai-org/GLM-4.7-Flash"
    },
    {
     "benchmark_id": "hle",
     "score": 14.4,
     "source_url": "https://huggingface.co/zai-org/GLM-4.7-Flash"
    },
    {
     "benchmark_id": "browsecomp",
     "score": 42.8,
     "source_url": "https://huggingface.co/zai-org/GLM-4.7-Flash"
    }
   ]
  },
  {
   "id": "glm-5",
   "name": "GLM-5",
   "org": "Z.ai",
   "release_date": "2026-02-12",
   "params": "744B-A40B",
   "license": "mit",
   "kind": "llm",
   "hf_id": "zai-org/GLM-5",
   "links": {
    "hf": "https://huggingface.co/zai-org/GLM-5",
    "paper": "https://arxiv.org/abs/2602.15763"
   },
   "notes": "Z.ai's frontier flagship, roughly double the scale of GLM-4.5, built for long-horizon agentic engineering with sparse attention over a 200K-token context.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 86,
     "source_url": "https://huggingface.co/zai-org/GLM-5"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 77.8,
     "source_url": "https://huggingface.co/zai-org/GLM-5"
    },
    {
     "benchmark_id": "hle",
     "score": 30.5,
     "source_url": "https://huggingface.co/zai-org/GLM-5"
    },
    {
     "benchmark_id": "browsecomp",
     "score": 62,
     "source_url": "https://huggingface.co/zai-org/GLM-5"
    }
   ],
   "external": [
    {
     "benchmark_id": "swe-bench",
     "score": 72.8,
     "platform": "swe-bench",
     "variant": "mini-SWE-agent + GLM-5 (high reasoning)",
     "source_url": "https://www.swebench.com"
    }
   ]
  },
  {
   "id": "glm-5-1",
   "name": "GLM-5.1",
   "org": "Z.ai",
   "release_date": "2026-04-07",
   "params": "754B",
   "license": "mit",
   "kind": "llm",
   "hf_id": "zai-org/GLM-5.1",
   "links": {
    "hf": "https://huggingface.co/zai-org/GLM-5.1"
   },
   "notes": "Flagship tuned to sustain performance across hundreds of rounds and thousands of tool calls, with stronger coding than GLM-5.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 86.2,
     "source_url": "https://huggingface.co/zai-org/GLM-5.1"
    },
    {
     "benchmark_id": "hle",
     "score": 31,
     "source_url": "https://huggingface.co/zai-org/GLM-5.1"
    },
    {
     "benchmark_id": "swe-bench-pro",
     "score": 58.4,
     "source_url": "https://huggingface.co/zai-org/GLM-5.1"
    },
    {
     "benchmark_id": "browsecomp",
     "score": 68,
     "source_url": "https://huggingface.co/zai-org/GLM-5.1"
    }
   ]
  },
  {
   "id": "glm-5-2",
   "name": "GLM-5.2",
   "org": "Z.ai",
   "release_date": "2026-06-16",
   "params": "753B",
   "license": "mit",
   "kind": "llm",
   "hf_id": "zai-org/GLM-5.2",
   "links": {
    "hf": "https://huggingface.co/zai-org/GLM-5.2",
    "blog": "https://huggingface.co/blog/zai-org/glm-52-blog"
   },
   "notes": "Long-horizon flagship adding a 1M-token context window through IndexShare sparse-attention reuse, aimed at sustained coding and agentic work.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 91.2,
     "source_url": "https://huggingface.co/zai-org/GLM-5.2"
    },
    {
     "benchmark_id": "hle",
     "score": 40.5,
     "source_url": "https://huggingface.co/zai-org/GLM-5.2"
    },
    {
     "benchmark_id": "swe-bench-pro",
     "score": 62.1,
     "source_url": "https://huggingface.co/zai-org/GLM-5.2"
    }
   ]
  },
  {
   "id": "gpt-4-1",
   "name": "GPT-4.1",
   "org": "OpenAI",
   "release_date": "2025-04-14",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://openai.com/index/gpt-4-1/"
   },
   "notes": "OpenAI's non-reasoning API model emphasizing coding, instruction-following and a 1-million-token context window.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 66.3,
     "source_url": "https://openai.com/index/gpt-4-1/"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 54.6,
     "source_url": "https://openai.com/index/gpt-4-1/"
    }
   ],
   "external": [
    {
     "benchmark_id": "swe-bench",
     "score": 39.6,
     "platform": "swe-bench",
     "variant": "mini-SWE-agent + GPT-4.1 (2025-04-14), verified by the maintainers",
     "source_url": "https://www.swebench.com"
    }
   ]
  },
  {
   "id": "gpt-4o",
   "name": "GPT-4o",
   "org": "OpenAI",
   "release_date": "2024-05-13",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://openai.com/index/hello-gpt-4o/"
   },
   "notes": "OpenAI's natively multimodal flagship model, processing text, audio, image and video in a single network. OpenAI publishes its benchmark results only as chart images, so no sourced scores are recorded here yet.",
   "lineage": [],
   "scores": [],
   "external": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 74,
     "platform": "artificial-analysis",
     "variant": "GPT-4o (May '24)",
     "source_url": "https://artificialanalysis.ai/models/gpt-4o-2024-05-13"
    },
    {
     "benchmark_id": "gpqa",
     "score": 52.6,
     "platform": "artificial-analysis",
     "variant": "GPT-4o (May '24)",
     "source_url": "https://artificialanalysis.ai/models/gpt-4o-2024-05-13"
    },
    {
     "benchmark_id": "hle",
     "score": 2.8,
     "platform": "artificial-analysis",
     "variant": "GPT-4o (May '24)",
     "source_url": "https://artificialanalysis.ai/models/gpt-4o-2024-05-13"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 21.6,
     "platform": "swe-bench",
     "variant": "mini-SWE-agent + GPT-4o (2024-11-20), verified by the maintainers",
     "source_url": "https://www.swebench.com"
    }
   ]
  },
  {
   "id": "gpt-5",
   "name": "GPT-5",
   "org": "OpenAI",
   "release_date": "2025-08-07",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://openai.com/index/introducing-gpt-5/"
   },
   "notes": "OpenAI's unified model system that routes between fast and deep-reasoning modes (Instant, Thinking, Pro).",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "aime",
     "score": 94.6,
     "source_url": "https://openai.com/index/introducing-gpt-5/"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 74.9,
     "source_url": "https://openai.com/index/introducing-gpt-5/"
    }
   ],
   "external": [
    {
     "benchmark_id": "swe-bench",
     "score": 65,
     "platform": "swe-bench",
     "variant": "mini-SWE-agent + GPT-5 (2025-08-07) (medium reasoning), verified by the maintainers",
     "source_url": "https://www.swebench.com"
    }
   ]
  },
  {
   "id": "gpt-5-1",
   "name": "GPT-5.1",
   "org": "OpenAI",
   "release_date": "2025-11-13",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://openai.com/index/gpt-5-1-for-developers/"
   },
   "notes": "Next iteration of the GPT-5 family with adaptive reasoning effort, including a no-reasoning fast mode, and an updated coding personality.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 88.1,
     "source_url": "https://openai.com/index/gpt-5-1-for-developers/"
    },
    {
     "benchmark_id": "aime",
     "score": 94,
     "source_url": "https://openai.com/index/gpt-5-1-for-developers/"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 76.3,
     "source_url": "https://openai.com/index/gpt-5-1-for-developers/"
    }
   ],
   "external": [
    {
     "benchmark_id": "swe-bench",
     "score": 66,
     "platform": "swe-bench",
     "variant": "mini-SWE-agent + GPT-5.1 (2025-11-13) (medium reasoning)",
     "source_url": "https://www.swebench.com"
    }
   ]
  },
  {
   "id": "gpt-5-1-codex-max",
   "name": "GPT-5.1-Codex-Max",
   "org": "OpenAI",
   "release_date": "2025-11-19",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://openai.com/index/gpt-5-1-codex-max/"
   },
   "notes": "Frontier agentic coding model built for long-running, project-scale software engineering, replacing GPT-5.1-Codex as the default Codex model.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "swe-bench",
     "score": 77.9,
     "source_url": "https://openai.com/index/gpt-5-1-codex-max/"
    }
   ]
  },
  {
   "id": "gpt-5-2",
   "name": "GPT-5.2",
   "org": "OpenAI",
   "release_date": "2025-12-11",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://openai.com/index/introducing-gpt-5-2/"
   },
   "notes": "Model series positioned for professional knowledge work, reporting state-of-the-art GDPval results alongside gains in coding, tool use and long context.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 92.4,
     "source_url": "https://openai.com/index/introducing-gpt-5-2/"
    },
    {
     "benchmark_id": "aime",
     "score": 100,
     "source_url": "https://openai.com/index/introducing-gpt-5-2/"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 80,
     "source_url": "https://openai.com/index/introducing-gpt-5-2/"
    }
   ],
   "external": [
    {
     "benchmark_id": "swe-bench",
     "score": 71.8,
     "platform": "swe-bench",
     "variant": "mini-SWE-agent + GPT-5.2 (2025-12-11) (high reasoning)",
     "source_url": "https://www.swebench.com"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 69,
     "platform": "swe-bench",
     "variant": "mini-SWE-agent + GPT-5.2 (2025-12-11)",
     "source_url": "https://www.swebench.com"
    }
   ]
  },
  {
   "id": "gpt-5-2-codex",
   "name": "GPT-5.2-Codex",
   "org": "OpenAI",
   "release_date": "2025-12-18",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://openai.com/index/introducing-gpt-5-2-codex/"
   },
   "notes": "Agentic coding model combining GPT-5.2's knowledge-work strengths with GPT-5.1-Codex-Max's terminal-using coding ability.",
   "lineage": [
    {
     "parent_id": "gpt-5-2",
     "relation": "merge"
    },
    {
     "parent_id": "gpt-5-1-codex-max",
     "relation": "merge"
    }
   ],
   "scores": [],
   "external": [
    {
     "benchmark_id": "gpqa",
     "score": 89.9,
     "platform": "artificial-analysis",
     "variant": "GPT-5.2 Codex (xhigh)",
     "source_url": "https://artificialanalysis.ai/models/gpt-5-2-codex"
    },
    {
     "benchmark_id": "hle",
     "score": 33.5,
     "platform": "artificial-analysis",
     "variant": "GPT-5.2 Codex (xhigh)",
     "source_url": "https://artificialanalysis.ai/models/gpt-5-2-codex"
    }
   ]
  },
  {
   "id": "gpt-5-3-codex",
   "name": "GPT-5.3-Codex",
   "org": "OpenAI",
   "release_date": "2026-02-05",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://openai.com/index/introducing-gpt-5-3-codex/"
   },
   "notes": "Advances GPT-5.2-Codex's agentic coding together with GPT-5.2's reasoning in one model; OpenAI states early versions helped debug its own training.",
   "lineage": [
    {
     "parent_id": "gpt-5-2-codex",
     "relation": "merge"
    },
    {
     "parent_id": "gpt-5-2",
     "relation": "merge"
    }
   ],
   "scores": [],
   "external": [
    {
     "benchmark_id": "gpqa",
     "score": 91.5,
     "platform": "artificial-analysis",
     "variant": "GPT-5.3 Codex (xhigh)",
     "source_url": "https://artificialanalysis.ai/models/gpt-5-3-codex"
    },
    {
     "benchmark_id": "hle",
     "score": 39.9,
     "platform": "artificial-analysis",
     "variant": "GPT-5.3 Codex (xhigh)",
     "source_url": "https://artificialanalysis.ai/models/gpt-5-3-codex"
    }
   ]
  },
  {
   "id": "gpt-5-4",
   "name": "GPT-5.4",
   "org": "OpenAI",
   "release_date": "2026-03-05",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://openai.com/index/introducing-gpt-5-4/"
   },
   "notes": "Frontier model for professional work with state-of-the-art coding, computer use, tool search and a 1M-token context window.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 92.8,
     "source_url": "https://openai.com/index/introducing-gpt-5-4/"
    }
   ]
  },
  {
   "id": "gpt-5-5",
   "name": "GPT-5.5",
   "org": "OpenAI",
   "release_date": "2026-04-24",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://openai.com/index/introducing-gpt-5-5/"
   },
   "notes": "Successor to GPT-5.4 across ChatGPT, the API and Codex, with reported gains in agentic coding, knowledge work and scientific research evaluations.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 93.6,
     "source_url": "https://openai.com/index/introducing-gpt-5-5/"
    },
    {
     "benchmark_id": "swe-bench-pro",
     "score": 59.4,
     "effort_note": "published in the GPT-5.6 launch table rather than its own, which is still OpenAI reporting on an OpenAI model",
     "source_url": "https://openai.com/index/gpt-5-6"
    }
   ]
  },
  {
   "id": "gpt-5-6",
   "name": "GPT-5.6",
   "org": "OpenAI",
   "release_date": "2026-07-09",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://openai.com/index/gpt-5-6/"
   },
   "notes": "Introduces durable capability tiers named Sol, Terra and Luna in place of the earlier Instant and Thinking naming. Scores are for the flagship Sol tier.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 94.6,
     "source_url": "https://openai.com/index/gpt-5-6/"
    },
    {
     "benchmark_id": "swe-bench-pro",
     "score": 64.6,
     "effort_note": "the Sol variant, which is the one this row tracks — OpenAI also lists Terra at 63.4 and Luna at 62.7",
     "source_url": "https://openai.com/index/gpt-5-6"
    },
    {
     "benchmark_id": "browsecomp",
     "score": 90.4,
     "effort_note": "the Sol variant, which is the one this row tracks — OpenAI also lists a Sol Ultra configuration at 92.2, Terra at 87.5 and Luna at 83.3",
     "source_url": "https://openai.com/index/gpt-5-6"
    }
   ],
   "external": [
    {
     "benchmark_id": "gpqa",
     "score": 94.1,
     "platform": "artificial-analysis",
     "variant": "GPT-5.6 Sol (max)",
     "source_url": "https://artificialanalysis.ai/evaluations/gpqa-diamond"
    }
   ]
  },
  {
   "id": "granite-4-1-30b",
   "name": "Granite 4.1 30B",
   "org": "IBM",
   "release_date": "2026-04-29",
   "params": "30B",
   "license": "apache-2.0",
   "kind": "llm",
   "hf_id": "ibm-granite/granite-4.1-30b",
   "links": {
    "hf": "https://huggingface.co/ibm-granite/granite-4.1-30b"
   },
   "notes": "Dense non-reasoning flagship of IBM's Granite 4.1 wave, positioned for enterprise generation, coding, retrieval and agentic tool calling.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 64.09,
     "source_url": "https://huggingface.co/ibm-granite/granite-4.1-30b"
    },
    {
     "benchmark_id": "gpqa",
     "score": 45.76,
     "source_url": "https://huggingface.co/ibm-granite/granite-4.1-30b"
    }
   ]
  },
  {
   "id": "grok-3",
   "name": "Grok 3",
   "org": "xAI",
   "release_date": "2025-02-19",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://x.ai/news/grok-3"
   },
   "notes": "xAI's flagship non-reasoning chat model, trained on the Colossus cluster.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 79.9,
     "source_url": "https://x.ai/news/grok-3"
    },
    {
     "benchmark_id": "gpqa",
     "score": 75.4,
     "source_url": "https://x.ai/news/grok-3"
    }
   ]
  },
  {
   "id": "grok-4",
   "name": "Grok 4",
   "org": "xAI",
   "release_date": "2025-07-09",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://x.ai/news/grok-4"
   },
   "notes": "xAI's reasoning-native flagship model, which uses tools during its chain of thought by default. xAI publishes its benchmark results only as chart images, so no sourced scores are recorded here yet.",
   "lineage": [],
   "scores": [],
   "external": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 86.6,
     "platform": "artificial-analysis",
     "variant": "Grok 4",
     "source_url": "https://artificialanalysis.ai/models/grok-4"
    },
    {
     "benchmark_id": "gpqa",
     "score": 87.7,
     "platform": "artificial-analysis",
     "variant": "Grok 4",
     "source_url": "https://artificialanalysis.ai/models/grok-4"
    },
    {
     "benchmark_id": "aime",
     "score": 92.7,
     "platform": "artificial-analysis",
     "variant": "Grok 4",
     "source_url": "https://artificialanalysis.ai/models/grok-4"
    },
    {
     "benchmark_id": "hle",
     "score": 23.9,
     "platform": "artificial-analysis",
     "variant": "Grok 4",
     "source_url": "https://artificialanalysis.ai/models/grok-4"
    }
   ]
  },
  {
   "id": "grok-4-1",
   "name": "Grok 4.1",
   "org": "xAI",
   "release_date": "2025-11-17",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://x.ai/news/grok-4-1"
   },
   "notes": "Update to Grok 4 focused on conversational usability, emotional intelligence and reduced hallucinations, in thinking and non-thinking configurations.",
   "lineage": [
    {
     "parent_id": "grok-4",
     "relation": "finetune"
    }
   ],
   "scores": []
  },
  {
   "id": "grok-4-1-fast",
   "name": "Grok 4.1 Fast",
   "org": "xAI",
   "release_date": "2025-11-19",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://x.ai/news/grok-4-1-fast"
   },
   "notes": "Tool-calling-optimised model with a 2M-token context window, launched alongside xAI's Agent Tools API.",
   "lineage": [],
   "scores": [],
   "external": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 85.4,
     "platform": "artificial-analysis",
     "variant": "Grok 4.1 Fast (Reasoning)",
     "source_url": "https://artificialanalysis.ai/models/grok-4-1-fast-reasoning"
    },
    {
     "benchmark_id": "gpqa",
     "score": 85.3,
     "platform": "artificial-analysis",
     "variant": "Grok 4.1 Fast (Reasoning)",
     "source_url": "https://artificialanalysis.ai/models/grok-4-1-fast-reasoning"
    },
    {
     "benchmark_id": "aime",
     "score": 89.3,
     "platform": "artificial-analysis",
     "variant": "Grok 4.1 Fast (Reasoning)",
     "source_url": "https://artificialanalysis.ai/models/grok-4-1-fast-reasoning"
    },
    {
     "benchmark_id": "hle",
     "score": 17.6,
     "platform": "artificial-analysis",
     "variant": "Grok 4.1 Fast (Reasoning)",
     "source_url": "https://artificialanalysis.ai/models/grok-4-1-fast-reasoning"
    },
    {
     "benchmark_id": "mmlu-pro",
     "score": 74.3,
     "platform": "artificial-analysis",
     "variant": "Grok 4.1 Fast (Non-reasoning)",
     "source_url": "https://artificialanalysis.ai/models/grok-4-1-fast"
    },
    {
     "benchmark_id": "gpqa",
     "score": 63.7,
     "platform": "artificial-analysis",
     "variant": "Grok 4.1 Fast (Non-reasoning)",
     "source_url": "https://artificialanalysis.ai/models/grok-4-1-fast"
    },
    {
     "benchmark_id": "aime",
     "score": 34.3,
     "platform": "artificial-analysis",
     "variant": "Grok 4.1 Fast (Non-reasoning)",
     "source_url": "https://artificialanalysis.ai/models/grok-4-1-fast"
    },
    {
     "benchmark_id": "hle",
     "score": 5,
     "platform": "artificial-analysis",
     "variant": "Grok 4.1 Fast (Non-reasoning)",
     "source_url": "https://artificialanalysis.ai/models/grok-4-1-fast"
    }
   ]
  },
  {
   "id": "grok-4-5",
   "name": "Grok 4.5",
   "org": "xAI",
   "release_date": "2026-07-16",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://x.ai/news/grok-4-5"
   },
   "notes": "xAI's flagship model trained alongside Cursor for coding, agentic and knowledge work, positioned as the default model in Grok Build.",
   "lineage": [],
   "scores": [],
   "external": [
    {
     "benchmark_id": "gpqa",
     "score": 93.1,
     "platform": "artificial-analysis",
     "variant": "Grok 4.5 (high)",
     "source_url": "https://artificialanalysis.ai/models/grok-4-5"
    },
    {
     "benchmark_id": "hle",
     "score": 40.3,
     "platform": "artificial-analysis",
     "variant": "Grok 4.5 (high)",
     "source_url": "https://artificialanalysis.ai/models/grok-4-5"
    }
   ]
  },
  {
   "hf_id": "NousResearch/Hermes-3-Llama-3.1-8B",
   "id": "hermes-3-llama-3-1-8b",
   "kind": "llm",
   "license": "llama-3.1",
   "lineage": [
    {
     "parent_id": "llama-3-1-8b",
     "relation": "finetune"
    }
   ],
   "links": {
    "hf": "https://huggingface.co/NousResearch/Hermes-3-Llama-3.1-8B"
   },
   "name": "Hermes 3 Llama 3.1 8B",
   "notes": "Community finetune of Llama 3.1 8B.",
   "org": "Nous Research",
   "params": "8B",
   "release_date": "2024-08-15",
   "scores": []
  },
  {
   "id": "hy3",
   "name": "Hy3",
   "org": "Tencent",
   "release_date": "2026-07-02",
   "params": "295B-A21B",
   "license": "apache-2.0",
   "kind": "llm",
   "hf_id": "tencent/Hy3",
   "links": {
    "hf": "https://huggingface.co/tencent/Hy3"
   },
   "notes": "Tencent's rebuilt Hunyuan line: a 295B mixture-of-experts with 21B active parameters, aimed at agentic coding and reasoning.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 90.4,
     "source_url": "https://huggingface.co/tencent/Hy3"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 78,
     "source_url": "https://huggingface.co/tencent/Hy3"
    },
    {
     "benchmark_id": "swe-bench-pro",
     "score": 57.9,
     "source_url": "https://huggingface.co/tencent/Hy3"
    },
    {
     "benchmark_id": "hle",
     "score": 53.2,
     "source_url": "https://huggingface.co/tencent/Hy3"
    }
   ]
  },
  {
   "id": "kimi-k2",
   "name": "Kimi K2",
   "org": "Moonshot AI",
   "release_date": "2025-07-11",
   "params": "1T-A32B",
   "license": "modified-mit",
   "kind": "llm",
   "hf_id": "moonshotai/Kimi-K2-Instruct",
   "links": {
    "hf": "https://huggingface.co/moonshotai/Kimi-K2-Instruct"
   },
   "notes": "A 1-trillion-parameter (32B active) mixture-of-experts model from Moonshot AI built and post-trained for agentic tool-use and coding tasks.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 81.1,
     "source_url": "https://huggingface.co/moonshotai/Kimi-K2-Instruct"
    },
    {
     "benchmark_id": "gpqa",
     "score": 75.1,
     "source_url": "https://huggingface.co/moonshotai/Kimi-K2-Instruct"
    },
    {
     "benchmark_id": "aime",
     "score": 49.5,
     "source_url": "https://huggingface.co/moonshotai/Kimi-K2-Instruct"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 65.8,
     "source_url": "https://huggingface.co/moonshotai/Kimi-K2-Instruct"
    }
   ]
  },
  {
   "id": "kimi-k2-5",
   "name": "Kimi K2.5",
   "org": "Moonshot AI",
   "release_date": "2026-01-27",
   "params": "1T-A32B",
   "license": "modified-mit",
   "kind": "llm",
   "hf_id": "moonshotai/Kimi-K2.5",
   "links": {
    "hf": "https://huggingface.co/moonshotai/Kimi-K2.5"
   },
   "notes": "Natively multimodal agentic model continually pretrained on roughly 15T mixed visual-text tokens, with an Agent Swarm mode that orchestrates up to 100 sub-agents per prompt.",
   "lineage": [
    {
     "parent_id": "kimi-k2",
     "relation": "finetune"
    }
   ],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 87.1,
     "source_url": "https://huggingface.co/moonshotai/Kimi-K2.5",
     "effort": "thinking",
     "effort_note": "thinking mode enabled"
    },
    {
     "benchmark_id": "gpqa",
     "score": 87.6,
     "source_url": "https://huggingface.co/moonshotai/Kimi-K2.5",
     "effort": "thinking",
     "effort_note": "thinking mode enabled"
    },
    {
     "benchmark_id": "aime",
     "score": 96.1,
     "source_url": "https://huggingface.co/moonshotai/Kimi-K2.5",
     "effort": "thinking",
     "effort_note": "thinking mode enabled"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 76.8,
     "source_url": "https://huggingface.co/moonshotai/Kimi-K2.5"
    },
    {
     "benchmark_id": "swe-bench-pro",
     "score": 50.7,
     "source_url": "https://huggingface.co/moonshotai/Kimi-K2.5"
    },
    {
     "benchmark_id": "hle",
     "score": 31.5,
     "source_url": "https://huggingface.co/moonshotai/Kimi-K2.5"
    },
    {
     "benchmark_id": "browsecomp",
     "score": 60.6,
     "source_url": "https://huggingface.co/moonshotai/Kimi-K2.5"
    }
   ],
   "external": [
    {
     "benchmark_id": "swe-bench",
     "score": 70.8,
     "platform": "swe-bench",
     "variant": "mini-SWE-agent + Kimi K2.5 (high reasoning)",
     "source_url": "https://www.swebench.com"
    }
   ]
  },
  {
   "id": "kimi-k2-6",
   "name": "Kimi K2.6",
   "org": "Moonshot AI",
   "release_date": "2026-04-20",
   "params": "1T-A32B",
   "license": "modified-mit",
   "kind": "llm",
   "hf_id": "moonshotai/Kimi-K2.6",
   "links": {
    "hf": "https://huggingface.co/moonshotai/Kimi-K2.6"
   },
   "notes": "Follow-up multimodal agentic model improving long-horizon coding, tool use and agentic browsing, with native INT4 quantisation.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 90.5,
     "source_url": "https://huggingface.co/moonshotai/Kimi-K2.6"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 80.2,
     "source_url": "https://huggingface.co/moonshotai/Kimi-K2.6"
    },
    {
     "benchmark_id": "swe-bench-pro",
     "score": 58.6,
     "source_url": "https://huggingface.co/moonshotai/Kimi-K2.6"
    },
    {
     "benchmark_id": "hle",
     "score": 36.4,
     "source_url": "https://huggingface.co/moonshotai/Kimi-K2.6"
    },
    {
     "benchmark_id": "browsecomp",
     "score": 83.2,
     "source_url": "https://huggingface.co/moonshotai/Kimi-K2.6"
    }
   ]
  },
  {
   "id": "kimi-k2-7-code",
   "name": "Kimi K2.7-Code",
   "org": "Moonshot AI",
   "release_date": "2026-06-12",
   "params": "1T-A32B",
   "license": "modified-mit",
   "kind": "llm",
   "hf_id": "moonshotai/Kimi-K2.7-Code",
   "links": {
    "hf": "https://huggingface.co/moonshotai/Kimi-K2.7-Code"
   },
   "notes": "Coding-specialised finetune of K2.6 for end-to-end software engineering, cutting thinking-token usage by roughly 30%.",
   "lineage": [
    {
     "parent_id": "kimi-k2-6",
     "relation": "finetune"
    }
   ],
   "scores": [],
   "external": [
    {
     "benchmark_id": "gpqa",
     "score": 89.6,
     "platform": "artificial-analysis",
     "variant": "Kimi K2.7 Code",
     "source_url": "https://artificialanalysis.ai/models/kimi-k2-7-code"
    },
    {
     "benchmark_id": "hle",
     "score": 32.8,
     "platform": "artificial-analysis",
     "variant": "Kimi K2.7 Code",
     "source_url": "https://artificialanalysis.ai/models/kimi-k2-7-code"
    }
   ]
  },
  {
   "hf_id": "meta-llama/Llama-3.1-405B",
   "id": "llama-3-1-405b",
   "kind": "llm",
   "license": "llama-3.1",
   "lineage": [],
   "links": {
    "hf": "https://huggingface.co/meta-llama/Llama-3.1-405B"
   },
   "name": "Llama 3.1 405B",
   "notes": "Meta's largest open-weights dense model.",
   "org": "Meta",
   "params": "405B",
   "release_date": "2024-07-23",
   "scores": [],
   "external": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 73.2,
     "platform": "artificial-analysis",
     "variant": "Llama 3.1 Instruct 405B",
     "source_url": "https://artificialanalysis.ai/models/llama-3-1-instruct-405b"
    },
    {
     "benchmark_id": "gpqa",
     "score": 51.5,
     "platform": "artificial-analysis",
     "variant": "Llama 3.1 Instruct 405B",
     "source_url": "https://artificialanalysis.ai/models/llama-3-1-instruct-405b"
    },
    {
     "benchmark_id": "aime",
     "score": 3,
     "platform": "artificial-analysis",
     "variant": "Llama 3.1 Instruct 405B",
     "source_url": "https://artificialanalysis.ai/models/llama-3-1-instruct-405b"
    },
    {
     "benchmark_id": "hle",
     "score": 4.2,
     "platform": "artificial-analysis",
     "variant": "Llama 3.1 Instruct 405B",
     "source_url": "https://artificialanalysis.ai/models/llama-3-1-instruct-405b"
    }
   ]
  },
  {
   "hf_id": "meta-llama/Llama-3.1-405B-Instruct",
   "id": "llama-3-1-405b-instruct",
   "kind": "llm",
   "license": "llama-3.1",
   "lineage": [
    {
     "parent_id": "llama-3-1-405b",
     "relation": "finetune"
    }
   ],
   "links": {
    "hf": "https://huggingface.co/meta-llama/Llama-3.1-405B-Instruct"
   },
   "name": "Llama 3.1 405B Instruct",
   "notes": "Instruction-tuned 405B.",
   "org": "Meta",
   "params": "405B",
   "release_date": "2024-07-23",
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 51.1,
     "source_url": "https://ai.meta.com/blog/meta-llama-3-1/"
    }
   ]
  },
  {
   "id": "llama-3-1-70b-instruct",
   "name": "Llama 3.1 70B Instruct",
   "org": "Meta",
   "release_date": "2024-07-23",
   "params": "70B",
   "license": "llama-3.1",
   "kind": "llm",
   "hf_id": "meta-llama/Llama-3.1-70B-Instruct",
   "links": {
    "hf": "https://huggingface.co/meta-llama/Llama-3.1-70B-Instruct",
    "blog": "https://ai.meta.com/blog/meta-llama-3-1/",
    "paper": "https://arxiv.org/abs/2407.21783"
   },
   "notes": "Meta's 70B-parameter multilingual instruction-tuned model from the Llama 3.1 family, positioned between the 8B and 405B variants.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 66.4,
     "source_url": "https://github.com/meta-llama/llama-models/blob/main/models/llama3_1/MODEL_CARD.md"
    },
    {
     "benchmark_id": "gpqa",
     "score": 48,
     "source_url": "https://github.com/meta-llama/llama-models/blob/main/models/llama3_3/MODEL_CARD.md"
    }
   ]
  },
  {
   "hf_id": "meta-llama/Llama-3.1-8B",
   "id": "llama-3-1-8b",
   "kind": "llm",
   "license": "llama-3.1",
   "lineage": [],
   "links": {
    "hf": "https://huggingface.co/meta-llama/Llama-3.1-8B"
   },
   "name": "Llama 3.1 8B",
   "notes": "Small sibling of the 3.1 family.",
   "org": "Meta",
   "params": "8B",
   "release_date": "2024-07-23",
   "scores": [],
   "external": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 47.6,
     "platform": "artificial-analysis",
     "variant": "Llama 3.1 Instruct 8B",
     "source_url": "https://artificialanalysis.ai/models/llama-3-1-instruct-8b"
    },
    {
     "benchmark_id": "gpqa",
     "score": 25.9,
     "platform": "artificial-analysis",
     "variant": "Llama 3.1 Instruct 8B",
     "source_url": "https://artificialanalysis.ai/models/llama-3-1-instruct-8b"
    },
    {
     "benchmark_id": "aime",
     "score": 4.3,
     "platform": "artificial-analysis",
     "variant": "Llama 3.1 Instruct 8B",
     "source_url": "https://artificialanalysis.ai/models/llama-3-1-instruct-8b"
    },
    {
     "benchmark_id": "hle",
     "score": 5.1,
     "platform": "artificial-analysis",
     "variant": "Llama 3.1 Instruct 8B",
     "source_url": "https://artificialanalysis.ai/models/llama-3-1-instruct-8b"
    }
   ]
  },
  {
   "id": "llama-3-1-8b-instruct",
   "name": "Llama 3.1 8B Instruct",
   "org": "Meta",
   "release_date": "2024-07-23",
   "params": "8B",
   "license": "llama-3.1",
   "kind": "llm",
   "hf_id": "meta-llama/Llama-3.1-8B-Instruct",
   "links": {
    "hf": "https://huggingface.co/meta-llama/Llama-3.1-8B-Instruct",
    "blog": "https://ai.meta.com/blog/meta-llama-3-1/",
    "paper": "https://arxiv.org/abs/2407.21783"
   },
   "notes": "Meta's 8B-parameter multilingual instruction-tuned model from the Llama 3.1 family, released alongside the 70B and 405B variants.",
   "lineage": [
    {
     "parent_id": "llama-3-1-8b",
     "relation": "finetune"
    }
   ],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 48.3,
     "source_url": "https://github.com/meta-llama/llama-models/blob/main/models/llama3_1/MODEL_CARD.md"
    },
    {
     "benchmark_id": "gpqa",
     "score": 31.8,
     "source_url": "https://github.com/meta-llama/llama-models/blob/main/models/llama3_3/MODEL_CARD.md"
    }
   ]
  },
  {
   "id": "llama-3-2-3b-instruct",
   "name": "Llama 3.2 3B Instruct",
   "org": "Meta",
   "release_date": "2024-10-24",
   "params": "3B",
   "license": "llama-3.2",
   "kind": "llm",
   "hf_id": "meta-llama/Llama-3.2-3B-Instruct",
   "links": {
    "hf": "https://huggingface.co/meta-llama/Llama-3.2-3B-Instruct",
    "blog": "https://ai.meta.com/blog/llama-3-2-connect-2024-vision-edge-mobile-devices/"
   },
   "notes": "A lightweight 3B-parameter instruction-tuned model created via pruning and distillation from Llama 3.1 8B, designed for on-device and edge deployment.",
   "lineage": [
    {
     "parent_id": "llama-3-1-8b",
     "relation": "distill"
    }
   ],
   "scores": [],
   "external": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 34.7,
     "platform": "artificial-analysis",
     "variant": "Llama 3.2 Instruct 3B",
     "source_url": "https://artificialanalysis.ai/models/llama-3-2-instruct-3b"
    },
    {
     "benchmark_id": "gpqa",
     "score": 22.4,
     "platform": "artificial-analysis",
     "variant": "Llama 3.2 Instruct 3B",
     "source_url": "https://artificialanalysis.ai/models/llama-3-2-instruct-3b"
    },
    {
     "benchmark_id": "aime",
     "score": 3.3,
     "platform": "artificial-analysis",
     "variant": "Llama 3.2 Instruct 3B",
     "source_url": "https://artificialanalysis.ai/models/llama-3-2-instruct-3b"
    },
    {
     "benchmark_id": "hle",
     "score": 5.2,
     "platform": "artificial-analysis",
     "variant": "Llama 3.2 Instruct 3B",
     "source_url": "https://artificialanalysis.ai/models/llama-3-2-instruct-3b"
    }
   ]
  },
  {
   "id": "llama-3-3-70b-instruct",
   "name": "Llama 3.3 70B Instruct",
   "org": "Meta",
   "release_date": "2024-12-06",
   "params": "70B",
   "license": "llama-3.3",
   "kind": "llm",
   "hf_id": "meta-llama/Llama-3.3-70B-Instruct",
   "links": {
    "hf": "https://huggingface.co/meta-llama/Llama-3.3-70B-Instruct",
    "blog": "https://ai.meta.com/blog/future-of-ai-built-with-llama/"
   },
   "notes": "A 70B-parameter instruction-tuned model that Meta positions as approaching Llama 3.1 405B's performance on several benchmarks at a fraction of the inference cost.",
   "lineage": [
    {
     "parent_id": "llama-3-1-70b-instruct",
     "relation": "finetune"
    }
   ],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 68.9,
     "source_url": "https://github.com/meta-llama/llama-models/blob/main/models/llama3_3/MODEL_CARD.md"
    },
    {
     "benchmark_id": "gpqa",
     "score": 50.5,
     "source_url": "https://github.com/meta-llama/llama-models/blob/main/models/llama3_3/MODEL_CARD.md"
    }
   ]
  },
  {
   "id": "llama-4-maverick",
   "name": "Llama 4 Maverick Instruct",
   "org": "Meta",
   "release_date": "2025-04-05",
   "params": "400B-A17B",
   "license": "llama-4",
   "kind": "llm",
   "hf_id": "meta-llama/Llama-4-Maverick-17B-128E-Instruct",
   "links": {
    "hf": "https://huggingface.co/meta-llama/Llama-4-Maverick-17B-128E-Instruct",
    "blog": "https://ai.meta.com/blog/llama-4-multimodal-intelligence/"
   },
   "notes": "A natively multimodal mixture-of-experts model with 17B active parameters routed across 128 experts (400B total), Meta's flagship Llama 4 release.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 80.5,
     "source_url": "https://github.com/meta-llama/llama-models/blob/main/models/llama4/MODEL_CARD.md"
    },
    {
     "benchmark_id": "gpqa",
     "score": 69.8,
     "source_url": "https://github.com/meta-llama/llama-models/blob/main/models/llama4/MODEL_CARD.md"
    }
   ],
   "external": [
    {
     "benchmark_id": "swe-bench",
     "score": 21,
     "platform": "swe-bench",
     "variant": "mini-SWE-agent + Llama 4 Maverick Instruct, verified by the maintainers",
     "source_url": "https://www.swebench.com"
    }
   ]
  },
  {
   "id": "llama-4-scout",
   "name": "Llama 4 Scout Instruct",
   "org": "Meta",
   "release_date": "2025-04-05",
   "params": "109B-A17B",
   "license": "llama-4",
   "kind": "llm",
   "hf_id": "meta-llama/Llama-4-Scout-17B-16E-Instruct",
   "links": {
    "hf": "https://huggingface.co/meta-llama/Llama-4-Scout-17B-16E-Instruct",
    "blog": "https://ai.meta.com/blog/llama-4-multimodal-intelligence/"
   },
   "notes": "Meta's first natively multimodal Llama, a 17B-active-parameter mixture-of-experts model (16 experts, 109B total) with a 10M-token context window.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 74.3,
     "source_url": "https://github.com/meta-llama/llama-models/blob/main/models/llama4/MODEL_CARD.md"
    },
    {
     "benchmark_id": "gpqa",
     "score": 57.2,
     "source_url": "https://github.com/meta-llama/llama-models/blob/main/models/llama4/MODEL_CARD.md"
    }
   ],
   "external": [
    {
     "benchmark_id": "swe-bench",
     "score": 9.1,
     "platform": "swe-bench",
     "variant": "mini-SWE-agent + Llama 4 Scout Instruct, verified by the maintainers",
     "source_url": "https://www.swebench.com"
    }
   ]
  },
  {
   "id": "magistral-medium",
   "name": "Magistral Medium",
   "org": "Mistral AI",
   "release_date": "2025-06-10",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://mistral.ai/news/magistral/",
    "paper": "https://arxiv.org/abs/2506.10910"
   },
   "notes": "Mistral AI's first dedicated reasoning model, trained with reinforcement learning on top of Mistral Medium 3; API and Le Chat only, no open weights.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 70.8,
     "source_url": "https://arxiv.org/abs/2506.10910"
    },
    {
     "benchmark_id": "aime",
     "score": 64.9,
     "source_url": "https://arxiv.org/abs/2506.10910"
    }
   ]
  },
  {
   "id": "minimax-m1",
   "name": "MiniMax M1",
   "org": "MiniMax",
   "release_date": "2025-06-17",
   "params": "456B-A45.9B",
   "license": "apache-2.0",
   "kind": "llm",
   "hf_id": "MiniMaxAI/MiniMax-M1-80k",
   "links": {
    "hf": "https://huggingface.co/MiniMaxAI/MiniMax-M1-80k",
    "paper": "https://arxiv.org/abs/2506.13585"
   },
   "notes": "Billed as the first open-weight large-scale hybrid-attention reasoning model, combining MoE with lightning attention for a native 1M-token context.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 81.1,
     "source_url": "https://github.com/MiniMax-AI/MiniMax-M1"
    },
    {
     "benchmark_id": "gpqa",
     "score": 70,
     "source_url": "https://github.com/MiniMax-AI/MiniMax-M1"
    },
    {
     "benchmark_id": "aime",
     "score": 76.9,
     "source_url": "https://github.com/MiniMax-AI/MiniMax-M1"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 56,
     "source_url": "https://github.com/MiniMax-AI/MiniMax-M1"
    }
   ]
  },
  {
   "id": "minimax-m2",
   "name": "MiniMax M2",
   "org": "MiniMax",
   "release_date": "2025-10-27",
   "params": "230B-A10B",
   "license": "modified-mit",
   "kind": "llm",
   "hf_id": "MiniMaxAI/MiniMax-M2",
   "links": {
    "hf": "https://huggingface.co/MiniMaxAI/MiniMax-M2"
   },
   "notes": "A 230B-parameter (10B active) MoE model that deliberately returned to full attention over M1's linear attention, optimized for coding and agentic workflows.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 82,
     "source_url": "https://huggingface.co/MiniMaxAI/MiniMax-M2"
    },
    {
     "benchmark_id": "gpqa",
     "score": 78,
     "source_url": "https://huggingface.co/MiniMaxAI/MiniMax-M2"
    },
    {
     "benchmark_id": "aime",
     "score": 78,
     "source_url": "https://huggingface.co/MiniMaxAI/MiniMax-M2"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 69.4,
     "source_url": "https://huggingface.co/MiniMaxAI/MiniMax-M2"
    },
    {
     "benchmark_id": "hle",
     "score": 12.5,
     "source_url": "https://huggingface.co/MiniMaxAI/MiniMax-M2"
    },
    {
     "benchmark_id": "browsecomp",
     "score": 44,
     "source_url": "https://huggingface.co/MiniMaxAI/MiniMax-M2"
    }
   ],
   "external": [
    {
     "benchmark_id": "swe-bench",
     "score": 61,
     "platform": "swe-bench",
     "variant": "mini-SWE-agent + Minimax M2, verified by the maintainers",
     "source_url": "https://www.swebench.com"
    }
   ]
  },
  {
   "id": "minimax-m2-1",
   "name": "MiniMax M2.1",
   "org": "MiniMax",
   "release_date": "2025-12-23",
   "params": "229B",
   "license": "modified-mit",
   "kind": "llm",
   "hf_id": "MiniMaxAI/MiniMax-M2.1",
   "links": {
    "hf": "https://huggingface.co/MiniMaxAI/MiniMax-M2.1",
    "blog": "https://www.minimax.io/news/minimax-m21"
   },
   "notes": "Incremental agentic-coding update to M2, focused on multilingual software development, tool use and long-horizon workflows.",
   "lineage": [
    {
     "parent_id": "minimax-m2",
     "relation": "finetune"
    }
   ],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 88,
     "source_url": "https://huggingface.co/MiniMaxAI/MiniMax-M2.1"
    },
    {
     "benchmark_id": "gpqa",
     "score": 83,
     "source_url": "https://huggingface.co/MiniMaxAI/MiniMax-M2.1"
    },
    {
     "benchmark_id": "aime",
     "score": 83,
     "source_url": "https://huggingface.co/MiniMaxAI/MiniMax-M2.1"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 74,
     "source_url": "https://huggingface.co/MiniMaxAI/MiniMax-M2.1"
    },
    {
     "benchmark_id": "hle",
     "score": 22.2,
     "source_url": "https://huggingface.co/MiniMaxAI/MiniMax-M2.1"
    },
    {
     "benchmark_id": "browsecomp",
     "score": 47.4,
     "source_url": "https://huggingface.co/MiniMaxAI/MiniMax-M2.1"
    }
   ]
  },
  {
   "id": "minimax-m3",
   "name": "MiniMax M3",
   "org": "MiniMax",
   "release_date": "2026-06-01",
   "params": "428B-A23B",
   "license": "minimax-community",
   "kind": "llm",
   "hf_id": "MiniMaxAI/MiniMax-M3",
   "links": {
    "hf": "https://huggingface.co/MiniMaxAI/MiniMax-M3",
    "blog": "https://www.minimax.io/models/text/m3"
   },
   "notes": "MiniMax's first natively multimodal open-weight flagship, introducing MiniMax Sparse Attention for a 1M-token context alongside frontier agentic coding.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "swe-bench",
     "score": 80.5,
     "source_url": "https://huggingface.co/MiniMaxAI/MiniMax-M3"
    },
    {
     "benchmark_id": "swe-bench-pro",
     "score": 59,
     "source_url": "https://huggingface.co/MiniMaxAI/MiniMax-M3"
    },
    {
     "benchmark_id": "browsecomp",
     "score": 83.5,
     "source_url": "https://huggingface.co/MiniMaxAI/MiniMax-M3"
    }
   ]
  },
  {
   "id": "ministral-3-14b",
   "name": "Ministral 3 14B Reasoning",
   "org": "Mistral AI",
   "release_date": "2025-12-02",
   "params": "14B",
   "license": "apache-2.0",
   "kind": "llm",
   "hf_id": "mistralai/Ministral-3-14B-Reasoning-2512",
   "links": {
    "hf": "https://huggingface.co/mistralai/Ministral-3-14B-Reasoning-2512",
    "blog": "https://mistral.ai/news/mistral-3/"
   },
   "notes": "Edge-sized open-weight reasoning model from the Mistral 3 launch, aimed at offline and on-device deployment.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "aime",
     "score": 85,
     "source_url": "https://mistral.ai/news/mistral-3/"
    }
   ]
  },
  {
   "id": "mistral-large-2",
   "name": "Mistral Large 2",
   "org": "Mistral AI",
   "release_date": "2024-07-24",
   "params": "123B",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "mistralai/Mistral-Large-Instruct-2407",
   "links": {
    "hf": "https://huggingface.co/mistralai/Mistral-Large-Instruct-2407",
    "blog": "https://mistral.ai/news/mistral-large-2407/"
   },
   "notes": "Mistral AI's flagship dense 123B model, released under the non-commercial Mistral Research License. The MMLU figure is reported for the pretrained checkpoint.",
   "lineage": [],
   "scores": [],
   "external": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 68.3,
     "platform": "artificial-analysis",
     "variant": "Mistral Large 2 (Jul '24)",
     "source_url": "https://artificialanalysis.ai/models/mistral-large-2407"
    },
    {
     "benchmark_id": "gpqa",
     "score": 47.2,
     "platform": "artificial-analysis",
     "variant": "Mistral Large 2 (Jul '24)",
     "source_url": "https://artificialanalysis.ai/models/mistral-large-2407"
    },
    {
     "benchmark_id": "aime",
     "score": 0,
     "platform": "artificial-analysis",
     "variant": "Mistral Large 2 (Jul '24)",
     "source_url": "https://artificialanalysis.ai/models/mistral-large-2407"
    },
    {
     "benchmark_id": "hle",
     "score": 3.2,
     "platform": "artificial-analysis",
     "variant": "Mistral Large 2 (Jul '24)",
     "source_url": "https://artificialanalysis.ai/models/mistral-large-2407"
    }
   ]
  },
  {
   "id": "mistral-large-3",
   "name": "Mistral Large 3",
   "org": "Mistral AI",
   "release_date": "2025-12-02",
   "params": "675B-A41B",
   "license": "apache-2.0",
   "kind": "llm",
   "hf_id": "mistralai/Mistral-Large-3-675B-Instruct-2512",
   "links": {
    "hf": "https://huggingface.co/mistralai/Mistral-Large-3-675B-Instruct-2512",
    "blog": "https://mistral.ai/news/mistral-3/"
   },
   "notes": "Mistral's flagship open-weight multimodal, multilingual granular mixture-of-experts model, the largest release in the Mistral 3 family.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 67.17,
     "source_url": "https://huggingface.co/mistralai/Mistral-Large-3-675B-Instruct-2512"
    }
   ]
  },
  {
   "id": "mistral-medium-3-5",
   "name": "Mistral Medium 3.5",
   "org": "Mistral AI",
   "release_date": "2026-04-28",
   "params": "128B",
   "license": "modified-mit",
   "kind": "llm",
   "hf_id": "mistralai/Mistral-Medium-3.5-128B",
   "links": {
    "hf": "https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"
   },
   "notes": "Dense open-weight frontier-class model optimised for agentic and coding use, previously an API-only tier.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "swe-bench",
     "score": 77.6,
     "source_url": "https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"
    }
   ]
  },
  {
   "id": "mixtral-8x22b",
   "name": "Mixtral 8x22B",
   "org": "Mistral AI",
   "release_date": "2024-04-17",
   "params": "141B-A39B",
   "license": "apache-2.0",
   "kind": "llm",
   "hf_id": "mistralai/Mixtral-8x22B-v0.1",
   "links": {
    "hf": "https://huggingface.co/mistralai/Mixtral-8x22B-v0.1",
    "blog": "https://mistral.ai/news/mixtral-8x22b/"
   },
   "notes": "Mistral AI's second open sparse mixture-of-experts model, using only 39B of its 141B parameters per token. Scores are for the base model.",
   "lineage": [],
   "scores": [],
   "external": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 53.7,
     "platform": "artificial-analysis",
     "variant": "Mixtral 8x22B Instruct",
     "source_url": "https://artificialanalysis.ai/models/mistral-8x22b-instruct"
    },
    {
     "benchmark_id": "gpqa",
     "score": 33.2,
     "platform": "artificial-analysis",
     "variant": "Mixtral 8x22B Instruct",
     "source_url": "https://artificialanalysis.ai/models/mistral-8x22b-instruct"
    },
    {
     "benchmark_id": "hle",
     "score": 4.1,
     "platform": "artificial-analysis",
     "variant": "Mixtral 8x22B Instruct",
     "source_url": "https://artificialanalysis.ai/models/mistral-8x22b-instruct"
    }
   ]
  },
  {
   "id": "nemotron-3-nano-30b-a3b",
   "name": "Nemotron 3 Nano 30B A3B",
   "org": "Nvidia",
   "release_date": "2025-12-15",
   "params": "30B-A3B",
   "license": "nvidia-open-model",
   "kind": "llm",
   "hf_id": "nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16",
   "links": {
    "hf": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16"
   },
   "notes": "Small hybrid Mamba-2 and Transformer mixture-of-experts reasoning model with a toggleable thinking mode and a 1M-token context.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 78.3,
     "source_url": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16"
    },
    {
     "benchmark_id": "aime",
     "score": 89.1,
     "source_url": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 38.8,
     "source_url": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16"
    },
    {
     "benchmark_id": "hle",
     "score": 10.6,
     "source_url": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16"
    }
   ]
  },
  {
   "id": "nemotron-3-super-120b-a12b",
   "name": "Nemotron 3 Super 120B A12B",
   "org": "Nvidia",
   "release_date": "2026-03-11",
   "params": "120B-A12B",
   "license": "nvidia-open-model",
   "kind": "llm",
   "hf_id": "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
   "links": {
    "hf": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16"
   },
   "notes": "Mid-size Nemotron 3 model using a latent mixture-of-experts hybrid with multi-token prediction, targeting multi-agent applications over a 1M-token context.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 83.73,
     "source_url": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16"
    },
    {
     "benchmark_id": "aime",
     "score": 90.21,
     "source_url": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 60.47,
     "source_url": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16"
    },
    {
     "benchmark_id": "hle",
     "score": 18.26,
     "source_url": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16"
    },
    {
     "benchmark_id": "browsecomp",
     "score": 31.28,
     "source_url": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16"
    }
   ]
  },
  {
   "id": "nemotron-3-ultra-550b-a55b",
   "name": "Nemotron 3 Ultra 550B A55B",
   "org": "Nvidia",
   "release_date": "2026-06-04",
   "params": "550B-A55B",
   "license": "openmdw-1.1",
   "kind": "llm",
   "hf_id": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16",
   "links": {
    "hf": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"
   },
   "notes": "Flagship of Nvidia's Nemotron 3 family, a hybrid Mamba, mixture-of-experts and attention model for frontier reasoning and long-horizon agentic work.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 86.8,
     "source_url": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 70.7,
     "source_url": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"
    },
    {
     "benchmark_id": "hle",
     "score": 26.7,
     "source_url": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"
    },
    {
     "benchmark_id": "browsecomp",
     "score": 44.4,
     "source_url": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"
    }
   ]
  },
  {
   "id": "o1",
   "name": "o1",
   "org": "OpenAI",
   "release_date": "2024-12-05",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://openai.com/index/openai-o1-system-card/"
   },
   "notes": "OpenAI's first publicly released reasoning model, trained to use an internal chain of thought before answering.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 78.3,
     "source_url": "https://cdn.openai.com/o1-system-card-20241205.pdf"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 40.9,
     "source_url": "https://cdn.openai.com/o1-system-card.pdf"
    }
   ]
  },
  {
   "id": "o3",
   "name": "o3",
   "org": "OpenAI",
   "release_date": "2025-04-16",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://openai.com/index/introducing-o3-and-o4-mini/"
   },
   "notes": "OpenAI's flagship reasoning model succeeding o1. Its announcement reports only tool-augmented figures, so no comparable no-tools scores are recorded here yet.",
   "lineage": [],
   "scores": [],
   "external": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 85.3,
     "platform": "artificial-analysis",
     "variant": "o3",
     "source_url": "https://artificialanalysis.ai/models/o3"
    },
    {
     "benchmark_id": "gpqa",
     "score": 82.7,
     "platform": "artificial-analysis",
     "variant": "o3",
     "source_url": "https://artificialanalysis.ai/models/o3"
    },
    {
     "benchmark_id": "aime",
     "score": 88.3,
     "platform": "artificial-analysis",
     "variant": "o3",
     "source_url": "https://artificialanalysis.ai/models/o3"
    },
    {
     "benchmark_id": "hle",
     "score": 20,
     "platform": "artificial-analysis",
     "variant": "o3",
     "source_url": "https://artificialanalysis.ai/models/o3"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 58.4,
     "platform": "swe-bench",
     "variant": "mini-SWE-agent + o3 (2025-04-16), verified by the maintainers",
     "source_url": "https://www.swebench.com"
    }
   ]
  },
  {
   "id": "o4-mini",
   "name": "o4-mini",
   "org": "OpenAI",
   "release_date": "2025-04-16",
   "params": "",
   "license": "proprietary",
   "kind": "llm",
   "hf_id": "",
   "links": {
    "blog": "https://openai.com/index/introducing-o3-and-o4-mini/"
   },
   "notes": "OpenAI's smaller, cost-efficient reasoning model released alongside o3, with native tool use and image reasoning. Its announcement reports only tool-augmented figures, so no comparable no-tools scores are recorded here yet.",
   "lineage": [],
   "scores": [],
   "external": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 83.2,
     "platform": "artificial-analysis",
     "variant": "o4-mini (high)",
     "source_url": "https://artificialanalysis.ai/models/o4-mini"
    },
    {
     "benchmark_id": "gpqa",
     "score": 78.4,
     "platform": "artificial-analysis",
     "variant": "o4-mini (high)",
     "source_url": "https://artificialanalysis.ai/models/o4-mini"
    },
    {
     "benchmark_id": "aime",
     "score": 90.7,
     "platform": "artificial-analysis",
     "variant": "o4-mini (high)",
     "source_url": "https://artificialanalysis.ai/models/o4-mini"
    },
    {
     "benchmark_id": "hle",
     "score": 17.5,
     "platform": "artificial-analysis",
     "variant": "o4-mini (high)",
     "source_url": "https://artificialanalysis.ai/models/o4-mini"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 45,
     "platform": "swe-bench",
     "variant": "mini-SWE-agent + o4-mini (2025-04-16), verified by the maintainers",
     "source_url": "https://www.swebench.com"
    }
   ]
  },
  {
   "id": "olmo-3-32b-think",
   "name": "Olmo 3-Think 32B",
   "org": "Ai2",
   "release_date": "2025-11-20",
   "params": "32B",
   "license": "apache-2.0",
   "kind": "llm",
   "hf_id": "allenai/Olmo-3-32B-Think",
   "links": {
    "hf": "https://huggingface.co/allenai/Olmo-3-32B-Think",
    "blog": "https://allenai.org/blog/olmo3"
   },
   "notes": "Ai2's flagship fully-open reasoning model, released with its complete training data, code and intermediate checkpoints under Apache 2.0.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "aime",
     "score": 72.5,
     "source_url": "https://huggingface.co/allenai/Olmo-3-32B-Think"
    }
   ]
  },
  {
   "id": "ornith-1-0-35b",
   "name": "Ornith-1.0-35B",
   "org": "DeepReinforce",
   "release_date": "2026-06-21",
   "params": "35B",
   "license": "mit",
   "kind": "llm",
   "hf_id": "deepreinforce-ai/Ornith-1.0-35B",
   "links": {
    "hf": "https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B"
   },
   "notes": "Agentic coding model from DeepReinforce, MIT licensed and reported on coding benchmarks only.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "swe-bench",
     "score": 75.6,
     "source_url": "https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B"
    },
    {
     "benchmark_id": "swe-bench-pro",
     "score": 50.4,
     "source_url": "https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B"
    }
   ]
  },
  {
   "id": "ornith-1-0-9b",
   "name": "Ornith-1.0-9B",
   "org": "DeepReinforce",
   "release_date": "2026-06-21",
   "params": "9B",
   "license": "mit",
   "kind": "llm",
   "hf_id": "deepreinforce-ai/Ornith-1.0-9B",
   "links": {
    "hf": "https://huggingface.co/deepreinforce-ai/Ornith-1.0-9B"
   },
   "notes": "Small sibling of Ornith-1.0-35B, same agentic coding focus at a size that fits one consumer card.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "swe-bench",
     "score": 69.4,
     "source_url": "https://huggingface.co/deepreinforce-ai/Ornith-1.0-9B"
    },
    {
     "benchmark_id": "swe-bench-pro",
     "score": 42.9,
     "source_url": "https://huggingface.co/deepreinforce-ai/Ornith-1.0-9B"
    }
   ]
  },
  {
   "id": "qwen2-5-72b-instruct",
   "name": "Qwen2.5 72B Instruct",
   "org": "Alibaba",
   "release_date": "2024-09-19",
   "params": "72B",
   "license": "qwen",
   "kind": "llm",
   "hf_id": "Qwen/Qwen2.5-72B-Instruct",
   "links": {
    "hf": "https://huggingface.co/Qwen/Qwen2.5-72B-Instruct",
    "blog": "https://qwenlm.github.io/blog/qwen2.5/",
    "paper": "https://arxiv.org/abs/2412.15115"
   },
   "notes": "Alibaba's flagship dense instruction-tuned model in the Qwen2.5 series, which spans 0.5B to 72B parameters.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 71.1,
     "source_url": "https://arxiv.org/html/2412.15115v1"
    },
    {
     "benchmark_id": "gpqa",
     "score": 49,
     "source_url": "https://arxiv.org/html/2412.15115v1"
    }
   ]
  },
  {
   "id": "qwen2-5-coder-32b-instruct",
   "name": "Qwen2.5-Coder 32B Instruct",
   "org": "Alibaba",
   "release_date": "2024-11-12",
   "params": "32B",
   "license": "apache-2.0",
   "kind": "llm",
   "hf_id": "Qwen/Qwen2.5-Coder-32B-Instruct",
   "links": {
    "hf": "https://huggingface.co/Qwen/Qwen2.5-Coder-32B-Instruct",
    "blog": "https://qwenlm.github.io/blog/qwen2.5-coder-family/",
    "paper": "https://arxiv.org/abs/2409.12186"
   },
   "notes": "Code-specialized member of the Qwen2.5 family that Alibaba positions as matching GPT-4o's coding ability among open models.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 62.3,
     "source_url": "https://arxiv.org/html/2409.12186v2"
    },
    {
     "benchmark_id": "gpqa",
     "score": 41.8,
     "source_url": "https://arxiv.org/html/2409.12186v2"
    }
   ],
   "external": [
    {
     "benchmark_id": "swe-bench",
     "score": 9,
     "platform": "swe-bench",
     "variant": "mini-SWE-agent + Qwen2.5-Coder 32B Instruct, verified by the maintainers",
     "source_url": "https://www.swebench.com"
    }
   ]
  },
  {
   "id": "qwen3-235b-a22b",
   "name": "Qwen3 235B A22B",
   "org": "Alibaba",
   "release_date": "2025-04-29",
   "params": "235B-A22B",
   "license": "apache-2.0",
   "kind": "llm",
   "hf_id": "Qwen/Qwen3-235B-A22B",
   "links": {
    "hf": "https://huggingface.co/Qwen/Qwen3-235B-A22B",
    "blog": "https://qwenlm.github.io/blog/qwen3/",
    "paper": "https://arxiv.org/abs/2505.09388"
   },
   "notes": "Qwen3's flagship mixture-of-experts model with a switchable thinking mode; scores reflect thinking mode.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 71.1,
     "source_url": "https://arxiv.org/pdf/2505.09388"
    },
    {
     "benchmark_id": "aime",
     "score": 81.5,
     "source_url": "https://arxiv.org/pdf/2505.09388"
    }
   ]
  },
  {
   "id": "qwen3-32b",
   "name": "Qwen3 32B",
   "org": "Alibaba",
   "release_date": "2025-04-29",
   "params": "32B",
   "license": "apache-2.0",
   "kind": "llm",
   "hf_id": "Qwen/Qwen3-32B",
   "links": {
    "hf": "https://huggingface.co/Qwen/Qwen3-32B",
    "blog": "https://qwenlm.github.io/blog/qwen3/",
    "paper": "https://arxiv.org/abs/2505.09388"
   },
   "notes": "Qwen3's flagship dense model with a switchable thinking mode; scores reflect thinking mode.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "gpqa",
     "score": 68.4,
     "source_url": "https://arxiv.org/pdf/2505.09388"
    },
    {
     "benchmark_id": "aime",
     "score": 72.9,
     "source_url": "https://arxiv.org/pdf/2505.09388"
    }
   ]
  },
  {
   "id": "qwen3-5-397b-a17b",
   "name": "Qwen3.5 397B A17B",
   "org": "Alibaba",
   "release_date": "2026-02-16",
   "params": "397B-A17B",
   "license": "apache-2.0",
   "kind": "llm",
   "hf_id": "Qwen/Qwen3.5-397B-A17B",
   "links": {
    "hf": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B",
    "blog": "https://qwen.ai/blog?id=qwen3.5"
   },
   "notes": "Alibaba's natively multimodal flagship combining Gated DeltaNet with a sparse mixture of experts, released open-weight under Apache 2.0.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 87.8,
     "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"
    },
    {
     "benchmark_id": "gpqa",
     "score": 88.4,
     "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 76.4,
     "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"
    },
    {
     "benchmark_id": "browsecomp",
     "score": 69,
     "source_url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B"
    }
   ]
  },
  {
   "id": "qwen3-6-27b",
   "name": "Qwen3.6-27B",
   "org": "Alibaba",
   "release_date": "2026-04-21",
   "params": "27B",
   "license": "apache-2.0",
   "kind": "llm",
   "hf_id": "Qwen/Qwen3.6-27B",
   "links": {
    "hf": "https://huggingface.co/Qwen/Qwen3.6-27B"
   },
   "notes": "Dense 27B pitched at flagship-level coding, thinking mode on by default, 262K context extensible to about 1M.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 86.2,
     "source_url": "https://huggingface.co/Qwen/Qwen3.6-27B",
     "effort": "thinking",
     "effort_note": "thinking mode on by default"
    },
    {
     "benchmark_id": "gpqa",
     "score": 87.8,
     "source_url": "https://huggingface.co/Qwen/Qwen3.6-27B",
     "effort": "thinking",
     "effort_note": "thinking mode on by default"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 77.2,
     "source_url": "https://huggingface.co/Qwen/Qwen3.6-27B",
     "effort": "thinking",
     "effort_note": "thinking mode on by default"
    },
    {
     "benchmark_id": "swe-bench-pro",
     "score": 53.5,
     "source_url": "https://huggingface.co/Qwen/Qwen3.6-27B",
     "effort": "thinking",
     "effort_note": "thinking mode on by default"
    }
   ]
  },
  {
   "id": "qwen3-6-35b-a3b",
   "name": "Qwen3.6-35B-A3B",
   "org": "Alibaba",
   "release_date": "2026-04-15",
   "params": "35B-A3B",
   "license": "apache-2.0",
   "kind": "llm",
   "hf_id": "Qwen/Qwen3.6-35B-A3B",
   "links": {
    "hf": "https://huggingface.co/Qwen/Qwen3.6-35B-A3B"
   },
   "notes": "Sparse sibling of Qwen3.6: 35B total with 3B active per token, thinking mode on by default.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "mmlu-pro",
     "score": 85.2,
     "source_url": "https://huggingface.co/Qwen/Qwen3.6-35B-A3B",
     "effort": "thinking",
     "effort_note": "thinking mode on by default"
    },
    {
     "benchmark_id": "gpqa",
     "score": 86,
     "source_url": "https://huggingface.co/Qwen/Qwen3.6-35B-A3B",
     "effort": "thinking",
     "effort_note": "thinking mode on by default"
    },
    {
     "benchmark_id": "swe-bench",
     "score": 73.4,
     "source_url": "https://huggingface.co/Qwen/Qwen3.6-35B-A3B",
     "effort": "thinking",
     "effort_note": "thinking mode on by default"
    },
    {
     "benchmark_id": "swe-bench-pro",
     "score": 49.5,
     "source_url": "https://huggingface.co/Qwen/Qwen3.6-35B-A3B",
     "effort": "thinking",
     "effort_note": "thinking mode on by default"
    },
    {
     "benchmark_id": "hle",
     "score": 21.4,
     "source_url": "https://huggingface.co/Qwen/Qwen3.6-35B-A3B",
     "effort": "thinking",
     "effort_note": "thinking mode on by default"
    }
   ]
  },
  {
   "id": "qwen3-coder-next",
   "name": "Qwen3-Coder-Next",
   "org": "Alibaba",
   "release_date": "2026-01-30",
   "params": "80B-A3B",
   "license": "apache-2.0",
   "kind": "llm",
   "hf_id": "Qwen/Qwen3-Coder-Next",
   "links": {
    "hf": "https://huggingface.co/Qwen/Qwen3-Coder-Next"
   },
   "notes": "Open-weight model built for coding agents: 80B total, 3B active per token, 256K context.",
   "lineage": [],
   "scores": [
    {
     "benchmark_id": "swe-bench",
     "score": 70.6,
     "source_url": "https://huggingface.co/Qwen/Qwen3-Coder-Next"
    },
    {
     "benchmark_id": "swe-bench-pro",
     "score": 44.3,
     "source_url": "https://huggingface.co/Qwen/Qwen3-Coder-Next"
    }
   ]
  }
 ]
}