{
  "schema_version": "1.0",
  "retrieved_at": "2026-09-08T21:21:19.228Z",
  "publisher": "TheQuery",
  "methodology": "https://www.thequery.in/research#methodology",
  "description": "Compiled reported evaluations, not independent tests by TheQuery. Null fields are unknown. Preserve evaluation conditions when comparing results.",
  "models": [
    {
      "slug": "claude-fable-5",
      "name": "Claude Fable 5",
      "developer": "Anthropic",
      "release_date": "2026-06-09T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Anthropic",
        "API access": "Yes",
        "Max output": "128K tokens",
        "Text input": "Yes",
        "Audio input": "No native audio input documented",
        "Text output": "Yes",
        "Video input": "No native video input documented",
        "API model ID": "claude-fable-5",
        "Audio output": "No",
        "Computer use": "Yes",
        "Image output": "No",
        "Release date": "2026-06-09",
        "Video output": "No",
        "Context window": "1M tokens",
        "Product access": "Claude + Claude Code + Claude API + cloud partners",
        "Cache write / 1M": "$12.50 (5 min) / $20 (1 hr)",
        "Knowledge cutoff": "Jan 2026",
        "Cached input / 1M": "$1",
        "Input / 1M tokens": "$10",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$50",
        "Reasoning / effort": "Adaptive thinking (always on); high default",
        "Image / vision input": "Yes",
        "Batch / flex discount": "Batch API: 50% discount",
        "Long-context surcharge": "None — standard pricing through 1M context",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://www.anthropic.com/news/claude-fable-5-mythos-5",
          "title": "Claude Fable 5 launch"
        },
        {
          "url": "https://platform.claude.com/docs/en/models/fable-5/introducing-claude-fable-5-and-claude-mythos-5",
          "title": "Claude Fable 5 platform docs"
        },
        {
          "url": "https://platform.claude.com/docs/en/about-claude/pricing",
          "title": "Claude pricing"
        }
      ],
      "notes": "Released June 9, temporarily suspended June 12 under an export-control directive, and restored globally July 1. Current API status is active.",
      "verified_at": "2026-09-07T04:00:13.944Z"
    },
    {
      "slug": "claude-fable-5-1",
      "name": "Claude Fable 5.1",
      "developer": "Anthropic",
      "release_date": "2026-09-01T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "OSWorld": "77.9% partial / 41.7% strict (OSWorld 2.0; Aug 2026 task release)",
        "Developer": "Anthropic",
        "API access": "Yes",
        "Max output": "128K tokens",
        "Text input": "Yes",
        "Audio input": "No",
        "CursorBench": "73.4% (CursorBench 3.2.0)",
        "OSWorld 2.0": "77.9% partial / 41.7% strict",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "claude-fable-5-1",
        "Audio output": "No",
        "Computer use": "Yes",
        "HLE-Verified": "60.9% no tools / 65.0% with tools",
        "Image output": "No",
        "Release date": "2026-09-01",
        "Video output": "No",
        "Context window": "1M tokens",
        "Product access": "Claude API + Amazon Bedrock + Google Cloud + Microsoft Foundry + Claude Platform on AWS",
        "Terminal-Bench": "55.8% (Terminal-Bench 4.0; Anthropic launch eval)",
        "Cache write / 1M": "$12.50 (5m) / $20 (1h)",
        "Knowledge cutoff": "Jun 2026",
        "Cached input / 1M": "$0.25",
        "Input / 1M tokens": "$10",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$50",
        "Reasoning / effort": "Adaptive thinking (always on); default high; per-message effort",
        "Terminal-Bench 4.0": "55.8%",
        "Humanity's Last Exam": "60.9% no tools / 65.0% with tools",
        "Image / vision input": "Yes",
        "Batch / flex discount": "Batch API: 50% off input and output",
        "Tool / function calling": "Yes",
        "MCP / tool-use benchmark": "AutomationBench 31.4%"
      },
      "sources": [
        {
          "url": "https://platform.claude.com/docs/en/models/fable-5-1/overview",
          "title": "Claude Fable 5.1 model overview"
        },
        {
          "url": "https://platform.claude.com/docs/en/about-claude/pricing",
          "title": "Claude Platform pricing"
        }
      ],
      "notes": "Active latest Fable model. Anthropic lists a 1M context window, 128K max output, always-on adaptive thinking, and a June 2026 knowledge cutoff.",
      "verified_at": "2026-09-06T16:49:35.977Z"
    },
    {
      "slug": "claude-mythos-5",
      "name": "Claude Mythos 5",
      "developer": "Anthropic",
      "release_date": "2026-06-09T00:00:00.000Z",
      "access": "restricted",
      "comparison_data": {
        "Developer": "Anthropic",
        "API access": "Invite only — Claude API",
        "Max output": "128K",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "claude-mythos-5",
        "Audio output": "No",
        "Computer use": "Yes",
        "Image output": "No",
        "Release date": "2026-06-09",
        "Video output": "No",
        "Context window": "1M",
        "Product access": "Project Glasswing + Amazon Bedrock + Google Cloud + Microsoft Foundry for approved organizations",
        "Cache write / 1M": "$12.50 (5m) / $20 (1h)",
        "Knowledge cutoff": "Jan 2026",
        "Cached input / 1M": "$1",
        "Input / 1M tokens": "$10",
        "Weights / license": "Restricted proprietary — invite only",
        "Output / 1M tokens": "$50",
        "Reasoning / effort": "Adaptive thinking (always on); default high",
        "Image / vision input": "Yes",
        "Batch / flex discount": "Batch API: 50% discount on input and output",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://platform.claude.com/docs/en/models/mythos-5/overview",
          "title": "Claude Mythos 5 model overview"
        },
        {
          "url": "https://www.anthropic.com/news/claude-fable-5-mythos-5",
          "title": "Claude Fable 5 and Claude Mythos 5 launch"
        }
      ],
      "notes": "Invite-only Project Glasswing model. Anthropic states that Mythos 5 shares Claude Fable 5 specifications and pricing; capability rows inherited below are limited to ordinary evaluations and explicitly exclude OSWorld-family and AutomationBench results that may be safeguard-sensitive.",
      "verified_at": "2026-09-07T09:26:58.233Z"
    },
    {
      "slug": "claude-mythos-5-1",
      "name": "Claude Mythos 5.1",
      "developer": "Anthropic",
      "release_date": "2026-09-01T00:00:00.000Z",
      "access": "restricted",
      "comparison_data": {
        "Developer": "Anthropic",
        "API access": "Invite only — Claude API",
        "Max output": "128K",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "claude-mythos-5-1",
        "Audio output": "No",
        "Computer use": "Yes",
        "Image output": "No",
        "Release date": "2026-09-01",
        "Video output": "No",
        "Context window": "1M",
        "Product access": "Project Glasswing + Amazon Bedrock + Google Cloud + Microsoft Foundry for approved organizations",
        "Terminal-Bench": "60.9% (Terminal-Bench 4.0; safeguards differ from Fable 5.1)",
        "Cache write / 1M": "$12.50 (5m) / $20 (1h)",
        "Knowledge cutoff": "Jun 2026",
        "Cached input / 1M": "$0.25",
        "Input / 1M tokens": "$10",
        "Weights / license": "Restricted proprietary — invite only",
        "Output / 1M tokens": "$50",
        "Reasoning / effort": "Adaptive thinking (always on); default high; per-message effort",
        "Terminal-Bench 4.0": "60.9% (safeguards differ from Fable 5.1)",
        "Image / vision input": "Yes",
        "Batch / flex discount": "Batch API: 50% discount on input and output",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://platform.claude.com/docs/en/models/mythos-5-1/overview",
          "title": "Claude Mythos 5.1 model overview"
        },
        {
          "url": "https://www.anthropic.com/claude/mythos",
          "title": "Claude Mythos product page"
        },
        {
          "url": "https://platform.claude.com/docs/en/release-notes/overview",
          "title": "Claude Platform release notes"
        }
      ],
      "notes": "Current invite-only Mythos model. Anthropic states Mythos 5.1 has the same capabilities, specifications and pricing as Claude Fable 5.1 for Project Glasswing participants; safeguard-sensitive computer-use evaluations are not inherited.",
      "verified_at": "2026-09-07T09:26:58.233Z"
    },
    {
      "slug": "claude-mythos-preview",
      "name": "Claude Mythos Preview",
      "developer": "Anthropic",
      "release_date": "2026-04-07T00:00:00.000Z",
      "access": "restricted",
      "comparison_data": {
        "Developer": "Anthropic",
        "API access": "Restricted Project Glasswing access only",
        "Max output": "128K",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "claude-mythos-preview (Project Glasswing gated preview)",
        "Audio output": "No",
        "Computer use": "Yes — agentic search and computer-use evaluation support",
        "Image output": "No",
        "Release date": "2026-04-07",
        "Video output": "No",
        "Context window": "1M",
        "Product access": "Project Glasswing vetted organizations",
        "Weights / license": "Restricted proprietary research preview",
        "Reasoning / effort": "Adaptive thinking; standard capability evals use max effort",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://www.anthropic.com/glasswing",
          "title": "Project Glasswing"
        },
        {
          "url": "https://www.anthropic.com/research/claude-mythos-preview-system-card",
          "title": "Claude Mythos Preview System Card"
        }
      ],
      "notes": "General-purpose unreleased frontier model used in Project Glasswing. Anthropic did not make Mythos Preview generally available. Historical per-token pricing is intentionally omitted because current first-party docs do not publish a stable rate for the retired gated preview.",
      "verified_at": "2026-09-07T09:26:58.233Z"
    },
    {
      "slug": "claude-opus-4-6",
      "name": "Claude Opus 4.6",
      "developer": "Anthropic",
      "release_date": "2026-02-05T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Anthropic",
        "API access": "Yes (legacy model)",
        "Max output": "128K tokens",
        "Text input": "Yes",
        "Audio input": "No native audio input documented",
        "Text output": "Yes",
        "Video input": "No native video input documented",
        "API model ID": "claude-opus-4-6",
        "Audio output": "No",
        "Computer use": "Yes",
        "Image output": "No",
        "Release date": "2026-02-05",
        "Video output": "No",
        "Context window": "1M tokens",
        "Product access": "Claude + Claude Code + Claude API + cloud partners",
        "Cache write / 1M": "$6.25 (5 min) / $10 (1 hr)",
        "Knowledge cutoff": "May 2025",
        "Cached input / 1M": "$0.50",
        "Input / 1M tokens": "$5",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$25",
        "Reasoning / effort": "Adaptive thinking; extended thinking deprecated; high default",
        "Image / vision input": "Yes",
        "Batch / flex discount": "Batch API: 50% discount",
        "Long-context surcharge": "None — standard pricing through 1M context",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://www.anthropic.com/news/claude-opus-4-6",
          "title": "Introducing Claude Opus 4.6"
        },
        {
          "url": "https://platform.claude.com/docs/en/models/opus-4-6/overview",
          "title": "Claude Opus 4.6 platform docs"
        },
        {
          "url": "https://platform.claude.com/docs/en/about-claude/pricing",
          "title": "Claude pricing"
        }
      ],
      "notes": "Still available as a legacy model. This is the final Opus release in the older tokenizer generation.",
      "verified_at": "2026-09-07T03:42:40.561Z"
    },
    {
      "slug": "claude-opus-4-7",
      "name": "Claude Opus 4.7",
      "developer": "Anthropic",
      "release_date": "2026-04-16T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Anthropic",
        "API access": "Yes (legacy model)",
        "Max output": "128K tokens",
        "Text input": "Yes",
        "Audio input": "No native audio input documented",
        "Text output": "Yes",
        "Video input": "No native video input documented",
        "API model ID": "claude-opus-4-7",
        "Audio output": "No",
        "Computer use": "Yes",
        "Image output": "No",
        "Release date": "2026-04-16",
        "Video output": "No",
        "Context window": "1M tokens",
        "Product access": "Claude + Claude Code + Claude API + cloud partners",
        "Cache write / 1M": "$6.25 (5 min) / $10 (1 hr)",
        "Knowledge cutoff": "Jan 2026",
        "Cached input / 1M": "$0.50",
        "Input / 1M tokens": "$5",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$25",
        "Reasoning / effort": "Adaptive thinking; high default; xhigh supported",
        "Image / vision input": "Yes",
        "Batch / flex discount": "Batch API: 50% discount",
        "Long-context surcharge": "None — standard pricing through 1M context",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://www.anthropic.com/news/claude-opus-4-7",
          "title": "Introducing Claude Opus 4.7"
        },
        {
          "url": "https://platform.claude.com/docs/en/models/opus-4-7/overview",
          "title": "Claude Opus 4.7 platform docs"
        },
        {
          "url": "https://platform.claude.com/docs/en/about-claude/pricing",
          "title": "Claude pricing"
        }
      ],
      "notes": "Still available as a legacy model. Uses Anthropic's newer tokenizer generation introduced with Claude 4.7.",
      "verified_at": "2026-09-07T03:42:40.561Z"
    },
    {
      "slug": "claude-opus-4-8",
      "name": "Claude Opus 4.8",
      "developer": "Anthropic",
      "release_date": "2026-05-28T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Anthropic",
        "API access": "Yes (legacy model)",
        "Max output": "128K tokens",
        "Text input": "Yes",
        "Audio input": "No native audio input documented",
        "Text output": "Yes",
        "Video input": "No native video input documented",
        "API model ID": "claude-opus-4-8",
        "Audio output": "No",
        "Computer use": "Yes",
        "Image output": "No",
        "Release date": "2026-05-28",
        "Video output": "No",
        "Context window": "1M tokens",
        "Product access": "Claude + Claude Code + Claude API + cloud partners",
        "Cache write / 1M": "$6.25 (5 min) / $10 (1 hr)",
        "Knowledge cutoff": "Jan 2026",
        "Cached input / 1M": "$0.50",
        "Input / 1M tokens": "$5",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$25",
        "Reasoning / effort": "Adaptive thinking; high default",
        "Image / vision input": "Yes",
        "Batch / flex discount": "Batch API: 50% discount",
        "Long-context surcharge": "None — standard pricing through 1M context",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://www.anthropic.com/news/claude-opus-4-8",
          "title": "Introducing Claude Opus 4.8"
        },
        {
          "url": "https://platform.claude.com/docs/en/models/opus-4-8/overview",
          "title": "Claude Opus 4.8 platform docs"
        },
        {
          "url": "https://platform.claude.com/docs/en/about-claude/pricing",
          "title": "Claude pricing"
        }
      ],
      "notes": "Still available as a legacy model. Fast mode is supported; standard catalog pricing records the default API tier.",
      "verified_at": "2026-09-07T03:42:40.561Z"
    },
    {
      "slug": "claude-opus-5",
      "name": "Claude Opus 5",
      "developer": "Anthropic",
      "release_date": "2026-07-24T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Anthropic",
        "API access": "Yes",
        "Max output": "128K tokens",
        "Text input": "Yes",
        "Audio input": "No native audio input documented",
        "Text output": "Yes",
        "Video input": "No native video input documented",
        "API model ID": "claude-opus-5",
        "Audio output": "No",
        "Computer use": "Yes",
        "Image output": "No",
        "Release date": "2026-07-24",
        "Video output": "No",
        "Context window": "1M tokens",
        "Product access": "Claude + Claude Code + Claude API + cloud partners",
        "Cache write / 1M": "$6.25 (5 min) / $10 (1 hr)",
        "Knowledge cutoff": "May 2026",
        "Cached input / 1M": "$0.50",
        "Input / 1M tokens": "$5",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$25",
        "Reasoning / effort": "Adaptive thinking; high default",
        "Image / vision input": "Yes",
        "Batch / flex discount": "Batch API: 50% discount",
        "Long-context surcharge": "None — standard pricing through 1M context",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://www.anthropic.com/news/claude-opus-5",
          "title": "Introducing Claude Opus 5"
        },
        {
          "url": "https://platform.claude.com/docs/en/models/opus-5/whats-new-opus-5",
          "title": "Claude Opus 5 platform docs"
        },
        {
          "url": "https://platform.claude.com/docs/en/about-claude/pricing",
          "title": "Claude pricing"
        }
      ],
      "notes": "Current Opus model. Fast mode is also available at a higher per-token rate; standard catalog pricing records the default API tier.",
      "verified_at": "2026-09-07T03:42:40.561Z"
    },
    {
      "slug": "claude-sonnet-4-6",
      "name": "Claude Sonnet 4.6",
      "developer": "Anthropic",
      "release_date": "2026-02-17T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Anthropic",
        "API access": "Yes (legacy model)",
        "Max output": "128K tokens",
        "Text input": "Yes",
        "Audio input": "No native audio input documented",
        "Text output": "Yes",
        "Video input": "No native video input documented",
        "API model ID": "claude-sonnet-4-6",
        "Audio output": "No",
        "Computer use": "Yes; computer_20251124 tool generation",
        "Image output": "No",
        "Release date": "2026-02-17",
        "Video output": "No",
        "Context window": "1M tokens",
        "Product access": "Claude + Claude Code + Claude API + cloud partners",
        "Cache write / 1M": "$3.75 (5 min) / $6 (1 hr)",
        "Knowledge cutoff": "Aug 2025",
        "Cached input / 1M": "$0.30",
        "Input / 1M tokens": "$3",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$15",
        "Reasoning / effort": "Adaptive thinking; extended thinking deprecated; high default",
        "Image / vision input": "Yes",
        "Batch / flex discount": "Batch API: 50% discount",
        "Long-context surcharge": "None — standard pricing through 1M context",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://www.anthropic.com/news/claude-sonnet-4-6",
          "title": "Introducing Claude Sonnet 4.6"
        },
        {
          "url": "https://platform.claude.com/docs/en/models/sonnet-4-6/overview",
          "title": "Claude Sonnet 4.6 platform docs"
        },
        {
          "url": "https://platform.claude.com/docs/en/about-claude/pricing",
          "title": "Claude pricing"
        }
      ],
      "notes": "Still available as a legacy model. Uses the pre-4.7 tokenizer generation; later Anthropic comparison tables revised several agentic benchmark values for methodology consistency.",
      "verified_at": "2026-09-07T03:42:40.561Z"
    },
    {
      "slug": "claude-sonnet-5",
      "name": "Claude Sonnet 5",
      "developer": "Anthropic",
      "release_date": "2026-06-30T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Anthropic",
        "API access": "Yes",
        "Max output": "128K tokens",
        "Text input": "Yes",
        "Audio input": "No native audio input documented",
        "Text output": "Yes",
        "Video input": "No native video input documented",
        "API model ID": "claude-sonnet-5",
        "Audio output": "No",
        "Computer use": "Yes; stable computer toolset and browser use",
        "Image output": "No",
        "Release date": "2026-06-30",
        "Video output": "No",
        "Context window": "1M tokens",
        "Product access": "Claude + Claude Code + Claude API + cloud partners",
        "Cache write / 1M": "$2.50 (5 min) / $4 (1 hr)",
        "Knowledge cutoff": "Jan 2026",
        "Cached input / 1M": "$0.20",
        "Input / 1M tokens": "$2",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$10",
        "Reasoning / effort": "Adaptive thinking on by default; high default",
        "Image / vision input": "Yes",
        "Batch / flex discount": "Batch API: 50% discount",
        "Long-context surcharge": "None — standard pricing through 1M context",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://www.anthropic.com/news/claude-sonnet-5",
          "title": "Introducing Claude Sonnet 5"
        },
        {
          "url": "https://platform.claude.com/docs/en/docs/about-claude/models/whats-new-sonnet-5",
          "title": "What's new in Claude Sonnet 5"
        },
        {
          "url": "https://platform.claude.com/docs/en/about-claude/pricing",
          "title": "Claude pricing"
        }
      ],
      "notes": "The launch $2/$10 input/output rate was made permanent in August 2026. Sonnet 5 uses Anthropic's newer tokenizer and keeps adaptive thinking enabled by default.",
      "verified_at": "2026-09-07T03:42:40.561Z"
    },
    {
      "slug": "command-a-plus",
      "name": "Command A+",
      "developer": "Cohere",
      "release_date": "2026-05-20T00:00:00.000Z",
      "access": "open_source",
      "comparison_data": {
        "Developer": "Cohere",
        "API access": "Yes",
        "Max output": "64K",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "command-a-plus-05-2026",
        "Computer use": "Via agent/tool integrations",
        "Release date": "2026-05-20",
        "Context window": "128K",
        "Product access": "Cohere API + Model Vault + downloadable weights",
        "Terminal-Bench": "25% (Terminal-Bench Hard)",
        "Knowledge cutoff": "Apr 1, 2025",
        "Input / 1M tokens": "Free through Cohere API until rate limits; production via Model Vault",
        "Weights / license": "Open source — Apache 2.0",
        "Output / 1M tokens": "Free through Cohere API until rate limits; production via Model Vault",
        "Reasoning / effort": "Reasoning supported",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes",
        "MCP / tool-use benchmark": "τ² Telecom 85%"
      },
      "sources": [
        {
          "url": "https://docs.cohere.com/docs/command-a-plus",
          "title": "Command A+ model docs"
        },
        {
          "url": "https://cohere.com/blog/command-a-plus",
          "title": "Command A+ launch"
        }
      ],
      "notes": "Cohere's first sparse MoE Command model: 218B total / 25B active, optimized for reasoning, RAG, agentic workflows, multilingual work and multimodal documents.",
      "verified_at": "2026-09-07T09:28:38.262Z"
    },
    {
      "slug": "deepseek-v4-flash",
      "name": "DeepSeek V4 Flash",
      "developer": "DeepSeek",
      "release_date": "2026-04-24T00:00:00.000Z",
      "access": "open_weights",
      "comparison_data": {
        "Developer": "DeepSeek",
        "API access": "Yes",
        "Max output": "384K tokens",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "deepseek-v4-flash",
        "Audio output": "No",
        "Computer use": "Via agent harnesses; no native computer-use tool documented",
        "Image output": "No",
        "Release date": "2026-04-24",
        "Video output": "No",
        "Context window": "1M tokens",
        "Product access": "DeepSeek API + weights",
        "Terminal-Bench": "82.7% (DeepSeek Harness minimal mode, max effort)",
        "Cached input / 1M": "$0.007 off-peak / $0.014 peak cache hit",
        "Input / 1M tokens": "$0.22 off-peak / $0.44 peak cache miss",
        "Weights / license": "Open weights — MIT",
        "Output / 1M tokens": "$0.66 off-peak / $1.32 peak",
        "Reasoning / effort": "Thinking + non-thinking; low / high / max, high default",
        "Terminal-Bench 2.1": "82.7% (DeepSeek Harness minimal mode, max effort)",
        "Image / vision input": "No on the main endpoint",
        "Tool / function calling": "Yes",
        "MCP / tool-use benchmark": "Toolathlon 70.3%"
      },
      "sources": [
        {
          "url": "https://www.deepseek.com/en/news/v4-preview/",
          "title": "DeepSeek V4 Preview"
        },
        {
          "url": "https://api-docs.deepseek.com/updates",
          "title": "DeepSeek API updates"
        },
        {
          "url": "https://api-docs.deepseek.com/quick_start/pricing",
          "title": "DeepSeek API pricing"
        },
        {
          "url": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash",
          "title": "DeepSeek V4 Flash weights"
        }
      ],
      "notes": "Current hosted endpoint is the 0731 Flash revision. Vision is a separate experimental V4-Flash-Vision endpoint and is not attributed to this text endpoint.",
      "verified_at": "2026-09-07T03:17:35.772Z"
    },
    {
      "slug": "deepseek-v4-flash-vision-exp",
      "name": "DeepSeek V4 Flash Vision Exp",
      "developer": "DeepSeek",
      "release_date": "2026-08-21T00:00:00.000Z",
      "access": "open_source",
      "comparison_data": {
        "Developer": "DeepSeek",
        "API access": "Experimental API + Responses + Anthropic-compatible API",
        "Max output": "384K",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No native video input documented",
        "API model ID": "deepseek-v4-flash-vision-exp",
        "Computer use": "Via multimodal agent/tool harnesses",
        "Release date": "2026-08-21",
        "Context window": "1M",
        "Product access": "DeepSeek API + downloadable weights",
        "Cached input / 1M": "$0.007 off-peak / $0.014 peak",
        "Input / 1M tokens": "$0.22 off-peak / $0.44 peak cache miss",
        "Weights / license": "Open source — MIT",
        "Output / 1M tokens": "$0.66 off-peak / $1.32 peak",
        "Reasoning / effort": "Thinking + non-thinking; max effort used for official agent evals",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://api-docs.deepseek.com/news/news260821/",
          "title": "DeepSeek V4 Flash Vision release"
        },
        {
          "url": "https://api-docs.deepseek.com/quick_start/pricing/",
          "title": "DeepSeek models and pricing"
        },
        {
          "url": "https://api-docs.deepseek.com/guides/vision/",
          "title": "DeepSeek Vision guide"
        },
        {
          "url": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp",
          "title": "Official weights"
        }
      ],
      "notes": "Experimental multimodal V4 Flash derivative. Images are tokenized for billing at up to 384 tokens each and use V4 Flash pricing.",
      "verified_at": "2026-09-07T09:28:38.262Z"
    },
    {
      "slug": "deepseek-v4-pro",
      "name": "DeepSeek V4 Pro",
      "developer": "DeepSeek",
      "release_date": "2026-04-24T00:00:00.000Z",
      "access": "open_weights",
      "comparison_data": {
        "Developer": "DeepSeek",
        "API access": "Yes",
        "Max output": "384K tokens",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "deepseek-v4-pro",
        "Audio output": "No",
        "Computer use": "Via agent harnesses; no native computer-use tool documented",
        "Image output": "No",
        "Release date": "2026-04-24",
        "Video output": "No",
        "Context window": "1M tokens",
        "Product access": "DeepSeek API + weights",
        "Terminal-Bench": "87.9% (Terminal-Bench 2.1; DeepSeek Harness, max effort)",
        "Cached input / 1M": "$0.022 off-peak / $0.044 peak cache hit",
        "Input / 1M tokens": "$0.66 off-peak / $1.32 peak cache miss",
        "Weights / license": "Open weights — MIT",
        "Output / 1M tokens": "$1.98 off-peak / $3.96 peak",
        "Reasoning / effort": "Thinking + non-thinking; low / high / max, high default",
        "Terminal-Bench 2.1": "87.9% (DeepSeek Harness, max effort)",
        "Humanity's Last Exam": "42.7% no tools / 60.0% with tools",
        "Image / vision input": "No on the main endpoint",
        "Tool / function calling": "Yes",
        "MCP / tool-use benchmark": "Toolathlon 74.1%; AutomationBench 31.8%"
      },
      "sources": [
        {
          "url": "https://www.deepseek.com/en/news/v4-preview/",
          "title": "DeepSeek V4 Preview"
        },
        {
          "url": "https://api-docs.deepseek.com/updates",
          "title": "DeepSeek API updates"
        },
        {
          "url": "https://api-docs.deepseek.com/quick_start/pricing",
          "title": "DeepSeek API pricing"
        },
        {
          "url": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro",
          "title": "DeepSeek V4 Pro weights"
        }
      ],
      "notes": "Current hosted endpoint is the 0813 Pro revision. Benchmark evidence remains tied to the documented DeepSeek harness rather than being flattened with third-party runs.",
      "verified_at": "2026-09-07T03:17:35.772Z"
    },
    {
      "slug": "ernie-5-1",
      "name": "ERNIE 5.1",
      "developer": "Baidu",
      "release_date": "2026-05-09T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "AIME": "99.6% (AIME 2026 with tools)",
        "Developer": "Baidu",
        "API access": "Baidu Qianfan",
        "Max output": "64K",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "ERNIE-5.1",
        "Computer use": "Via Qianfan agent/tool integrations",
        "Release date": "2026-05-09",
        "Context window": "128K",
        "Product access": "ERNIE chat + Baidu AI Studio Playground + Qianfan API",
        "Input / 1M tokens": "¥4.00 ≤32K input / ¥6.00 >32K–128K",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "¥18.00 ≤32K input / ¥22.00 >32K–128K",
        "Reasoning / effort": "Agentic post-training + tool-augmented reasoning",
        "Image / vision input": "No native input documented for ERNIE 5.1 text generation",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://ernie.baidu.com/blog/posts/ernie-5.1-0508-release/",
          "title": "ERNIE 5.1 launch"
        },
        {
          "url": "https://cloud.baidu.com/doc/qianfan/s/wmh4sv6ya",
          "title": "Baidu Qianfan pricing"
        }
      ],
      "notes": "Text-generation model distilled from ERNIE 5.0's elastic pre-training matrix. Baidu reports roughly one-third the total parameters and half the active parameters of ERNIE 5.0, without publishing an absolute count.",
      "verified_at": "2026-09-07T09:28:38.262Z"
    },
    {
      "slug": "gemini-3-1-flash-image",
      "name": "Gemini 3.1 Flash Image",
      "developer": "Google DeepMind",
      "release_date": "2026-02-26T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Google DeepMind",
        "API access": "Yes",
        "Max output": "32K tokens",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "gemini-3.1-flash-image",
        "Audio output": "No",
        "Computer use": "No",
        "Image output": "Yes (0.5K/1K/2K/4K)",
        "Release date": "2026-02-26",
        "Video output": "No",
        "Context window": "128K tokens",
        "Product access": "Google AI Studio + Gemini API",
        "Knowledge cutoff": "January 2025",
        "Input / 1M tokens": "$0.50 text/image",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$3 text/thinking / $60 image tokens",
        "Reasoning / effort": "Thinking supported",
        "Image / vision input": "Yes",
        "Batch / flex discount": "Batch: $0.25 input; $1.50 text/thinking / $30 image tokens output",
        "Tool / function calling": "No"
      },
      "sources": [
        {
          "url": "https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-image",
          "title": "Gemini 3.1 Flash Image model docs"
        },
        {
          "url": "https://ai.google.dev/gemini-api/docs/pricing",
          "title": "Gemini API pricing"
        },
        {
          "url": "https://ai.google.dev/gemini-api/docs/deprecations",
          "title": "Gemini deprecations"
        }
      ],
      "notes": "Nano Banana 2 preview launched February 26; stable endpoint reached GA May 28, 2026.",
      "verified_at": "2026-09-06T16:49:37.841Z"
    },
    {
      "slug": "gemini-3-1-flash-lite",
      "name": "Gemini 3.1 Flash-Lite",
      "developer": "Google DeepMind",
      "release_date": "2026-03-03T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Google DeepMind",
        "API access": "Yes",
        "Max output": "64K tokens",
        "Text input": "Yes",
        "Audio input": "Yes",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "gemini-3.1-flash-lite",
        "Audio output": "No",
        "Computer use": "No",
        "Image output": "No",
        "Release date": "2026-03-03",
        "Video output": "No",
        "Context window": "1M tokens",
        "Product access": "Google AI Studio + Gemini API",
        "Knowledge cutoff": "Jan 2025",
        "Cached input / 1M": "$0.025 text/image/video / $0.05 audio",
        "Input / 1M tokens": "$0.25 text/image/video / $0.50 audio",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$1.50",
        "Reasoning / effort": "Thinking supported",
        "Image / vision input": "Yes",
        "Batch / flex discount": "Batch: $0.125 text/image/video, $0.25 audio input; $0.75 output",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-lite",
          "title": "Gemini 3.1 Flash-Lite model docs"
        },
        {
          "url": "https://ai.google.dev/gemini-api/docs/pricing",
          "title": "Gemini API pricing"
        },
        {
          "url": "https://ai.google.dev/gemini-api/docs/deprecations",
          "title": "Gemini deprecations"
        }
      ],
      "notes": "Preview launched March 3; stable endpoint reached GA May 7, 2026.",
      "verified_at": "2026-09-06T16:49:36.933Z"
    },
    {
      "slug": "gemini-3-1-flash-lite-image",
      "name": "Gemini 3.1 Flash-Lite Image",
      "developer": "Google DeepMind",
      "release_date": "2026-06-30T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Google DeepMind",
        "API access": "Yes",
        "Max output": "4K tokens",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "gemini-3.1-flash-lite-image",
        "Audio output": "No",
        "Computer use": "No",
        "Image output": "Yes (1K)",
        "Release date": "2026-06-30",
        "Video output": "No",
        "Context window": "64K tokens",
        "Product access": "Google AI Studio + Gemini API",
        "Input / 1M tokens": "$0.25 text/image/video",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$1.50 text/thinking / $30 image tokens",
        "Reasoning / effort": "minimal / high thinking",
        "Image / vision input": "Yes",
        "Batch / flex discount": "Batch: $0.125 input; $0.75 text/thinking / $15 image tokens output",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-lite-image",
          "title": "Gemini 3.1 Flash-Lite Image model docs"
        },
        {
          "url": "https://ai.google.dev/gemini-api/docs/pricing",
          "title": "Gemini API pricing"
        },
        {
          "url": "https://ai.google.dev/gemini-api/docs/changelog",
          "title": "Gemini API release notes"
        }
      ],
      "notes": "Nano Banana 2 Lite is the low-latency 1K image-generation/editing variant.",
      "verified_at": "2026-09-06T16:49:37.841Z"
    },
    {
      "slug": "gemini-3-1-flash-live",
      "name": "Gemini 3.1 Flash Live",
      "developer": "Google DeepMind",
      "release_date": "2026-03-26T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Google DeepMind",
        "API access": "Preview via Live API",
        "Max output": "64K tokens",
        "Text input": "Yes",
        "Audio input": "Yes",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "gemini-3.1-flash-live-preview",
        "Audio output": "Yes",
        "Computer use": "No native computer-use tool documented",
        "Image output": "No",
        "Release date": "2026-03-26",
        "Video output": "No",
        "Context window": "128K tokens",
        "Product access": "Google AI Studio + Gemini Live API",
        "Knowledge cutoff": "Jan 2025",
        "Input / 1M tokens": "$0.75 text / $3 audio / $1 image-video",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$4.50 text / $12 audio",
        "Reasoning / effort": "minimal / low / medium / high; minimal default",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes (synchronous)"
      },
      "sources": [
        {
          "url": "https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-live-preview",
          "title": "Gemini 3.1 Flash Live model docs"
        },
        {
          "url": "https://ai.google.dev/gemini-api/docs/pricing",
          "title": "Gemini API pricing"
        }
      ],
      "notes": "Low-latency audio-to-audio preview model; async function calling is not supported.",
      "verified_at": "2026-09-06T16:49:37.841Z"
    },
    {
      "slug": "gemini-3-1-flash-tts",
      "name": "Gemini 3.1 Flash TTS",
      "developer": "Google DeepMind",
      "release_date": "2026-04-15T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Google DeepMind",
        "API access": "Preview",
        "Max output": "16K tokens",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "No",
        "Video input": "No",
        "API model ID": "gemini-3.1-flash-tts-preview",
        "Audio output": "Yes",
        "Computer use": "No",
        "Image output": "No",
        "Release date": "2026-04-15",
        "Video output": "No",
        "Context window": "8K tokens",
        "Product access": "Google AI Studio + Gemini API",
        "Knowledge cutoff": "Jan 2025",
        "Input / 1M tokens": "$1 text",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$20 audio",
        "Reasoning / effort": "Not supported",
        "Image / vision input": "No",
        "Batch / flex discount": "Batch: $0.50 input / $10 audio output",
        "Tool / function calling": "No"
      },
      "sources": [
        {
          "url": "https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-tts-preview",
          "title": "Gemini 3.1 Flash TTS model docs"
        },
        {
          "url": "https://ai.google.dev/gemini-api/docs/pricing",
          "title": "Gemini API pricing"
        }
      ],
      "notes": null,
      "verified_at": "2026-09-06T16:49:37.841Z"
    },
    {
      "slug": "gemini-3-1-pro",
      "name": "Gemini 3.1 Pro",
      "developer": "Google DeepMind",
      "release_date": "2026-02-19T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Google DeepMind",
        "API access": "Preview",
        "Max output": "64K tokens",
        "Text input": "Yes",
        "Audio input": "Yes",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "gemini-3.1-pro-preview",
        "Audio output": "No",
        "Computer use": "No native computer-use tool documented",
        "Image output": "No",
        "Release date": "2026-02-19",
        "Video output": "No",
        "Context window": "1M tokens",
        "Product access": "Google AI Studio + Gemini API",
        "Knowledge cutoff": "Jan 2025",
        "Cached input / 1M": "$0.20 <=200K / $0.40 >200K",
        "Input / 1M tokens": "$2 <=200K / $4 >200K",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$12 <=200K / $18 >200K",
        "Reasoning / effort": "Thinking supported",
        "Image / vision input": "Yes",
        "Long-context surcharge": ">200K prompts use the higher input/output/cache tier",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://ai.google.dev/gemini-api/docs/models/gemini-3.1-pro-preview",
          "title": "Gemini 3.1 Pro Preview model docs"
        },
        {
          "url": "https://ai.google.dev/gemini-api/docs/pricing",
          "title": "Gemini API pricing"
        }
      ],
      "notes": "Preview model; custom-tools endpoint is also available for bash/custom-tool agent workflows.",
      "verified_at": "2026-09-06T16:49:36.933Z"
    },
    {
      "slug": "gemini-3-5-audio",
      "name": "Gemini 3.5 Audio",
      "developer": "Google DeepMind",
      "release_date": "2026-08-26T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Google DeepMind",
        "API access": "Yes / preview depending variant",
        "Max output": "64K Live Translate / 32K Transcribe",
        "Text input": "Yes",
        "Audio input": "Yes",
        "Text output": "Yes",
        "Video input": "No in documented Audio variants",
        "API model ID": "gemini-3.5-live-translate-preview / gemini-3.5-transcribe / gemini-3.5-transcribe-live",
        "Audio output": "Yes (Live Translate)",
        "Computer use": "No native computer-use tool documented",
        "Image output": "No",
        "Release date": "2026-08-26",
        "Video output": "No",
        "Context window": "128K Live Translate / 96K Transcribe",
        "Product access": "Gemini API + Google AI Studio; Transcribe variants also Gemini app/Vertex AI and other Google products",
        "Knowledge cutoff": "Jan 2025",
        "Input / 1M tokens": "Live Translate: $3.50 audio (~$0.0053/min)",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "Live Translate: $21 audio (~$0.0315/min)",
        "Image / vision input": "No in documented Audio variants",
        "Tool / function calling": "Variant-dependent"
      },
      "sources": [
        {
          "url": "https://deepmind.google/models/model-cards/gemini-3-5-audio/",
          "title": "Gemini 3.5 Audio model card"
        },
        {
          "url": "https://ai.google.dev/gemini-api/docs/pricing",
          "title": "Gemini API pricing"
        }
      ],
      "notes": "Catalog row groups Live Translate, Transcribe and Transcribe Live; capabilities and limits differ by variant.",
      "verified_at": "2026-09-06T16:49:37.841Z"
    },
    {
      "slug": "gemini-3-5-flash",
      "name": "Gemini 3.5 Flash",
      "developer": "Google DeepMind",
      "release_date": "2026-05-19T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Google DeepMind",
        "API access": "Yes",
        "Max output": "64K tokens",
        "Text input": "Yes",
        "Audio input": "Yes",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "gemini-3.5-flash",
        "Audio output": "No",
        "Computer use": "Yes (Preview)",
        "GDPval-AA v2": "1,656 Elo (Google launch evaluation)",
        "Image output": "No",
        "Release date": "2026-05-19",
        "Video output": "No",
        "Context window": "1M tokens",
        "Product access": "Gemini app + Google AI Studio + Gemini API + Antigravity + Gemini Enterprise",
        "Knowledge cutoff": "Jan 2025",
        "Cached input / 1M": "$0.15",
        "Input / 1M tokens": "$1.50",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$9",
        "Reasoning / effort": "minimal / low / medium / high thinking",
        "Terminal-Bench 2.1": "76.2%",
        "Image / vision input": "Yes",
        "Batch / flex discount": "Batch: $0.75 input / $4.50 output",
        "Tool / function calling": "Yes",
        "MCP / tool-use benchmark": "MCP Atlas 83.6%"
      },
      "sources": [
        {
          "url": "https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-5/",
          "title": "Gemini 3.5 Flash launch"
        },
        {
          "url": "https://ai.google.dev/gemini-api/docs/models/gemini-3.5-flash",
          "title": "Gemini 3.5 Flash model docs"
        },
        {
          "url": "https://ai.google.dev/gemini-api/docs/pricing",
          "title": "Gemini API pricing"
        },
        {
          "url": "https://deepmind.google/models/model-cards/gemini-3-6-flash/",
          "title": "Gemini 3.6 Flash model card"
        }
      ],
      "notes": null,
      "verified_at": "2026-09-06T16:49:36.933Z"
    },
    {
      "slug": "gemini-3-5-flash-lite",
      "name": "Gemini 3.5 Flash-Lite",
      "developer": "Google DeepMind",
      "release_date": "2026-07-21T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Google DeepMind",
        "API access": "Yes",
        "Max output": "64K tokens",
        "Text input": "Yes",
        "Audio input": "Yes",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "gemini-3.5-flash-lite",
        "Audio output": "No",
        "Computer use": "Yes (Preview)",
        "Image output": "No",
        "Release date": "2026-07-21",
        "Video output": "No",
        "Context window": "1M tokens",
        "Product access": "Google AI Studio + Gemini API",
        "Cached input / 1M": "$0.03",
        "Input / 1M tokens": "$0.30 text/image/video/audio",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$2.50",
        "Reasoning / effort": "Thinking supported",
        "Image / vision input": "Yes",
        "Batch / flex discount": "Batch/Flex: $0.15 input / $1.25 output",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://ai.google.dev/gemini-api/docs/models/gemini-3.5-flash-lite",
          "title": "Gemini 3.5 Flash-Lite model docs"
        },
        {
          "url": "https://ai.google.dev/gemini-api/docs/pricing",
          "title": "Gemini API pricing"
        },
        {
          "url": "https://ai.google.dev/gemini-api/docs/changelog",
          "title": "Gemini API release notes"
        }
      ],
      "notes": null,
      "verified_at": "2026-09-06T16:49:36.933Z"
    },
    {
      "slug": "gemini-3-6-flash",
      "name": "Gemini 3.6 Flash",
      "developer": "Google DeepMind",
      "release_date": "2026-07-21T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Google DeepMind",
        "API access": "Yes",
        "Max output": "64K tokens",
        "Text input": "Yes",
        "Audio input": "Yes",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "gemini-3.6-flash",
        "Audio output": "No",
        "Computer use": "Yes (Preview)",
        "Image output": "No",
        "Release date": "2026-07-21",
        "Video output": "No",
        "Context window": "1M tokens",
        "Product access": "Google AI Studio + Gemini API + Gemini Enterprise",
        "Cached input / 1M": "$0.075 introductory through Dec 31, 2026",
        "Input / 1M tokens": "$0.75 introductory through Dec 31, 2026",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$3.75 introductory through Dec 31, 2026",
        "Reasoning / effort": "Thinking supported",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://ai.google.dev/gemini-api/docs/models/gemini-3.6-flash",
          "title": "Gemini 3.6 Flash model docs"
        },
        {
          "url": "https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-6-flash-3-5-flash-lite-3-5-flash-cyber/",
          "title": "Gemini 3.6 Flash launch"
        },
        {
          "url": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
          "title": "Gemini 3.7 Flash model card"
        },
        {
          "url": "https://ai.google.dev/gemini-api/docs/pricing",
          "title": "Gemini API pricing"
        }
      ],
      "notes": "Current $0.75/$3.75 pricing is introductory through December 31, 2026; standard $1.50/$7.50 begins January 1, 2027.",
      "verified_at": "2026-09-06T16:49:36.933Z"
    },
    {
      "slug": "gemini-3-7-flash",
      "name": "Gemini 3.7 Flash",
      "developer": "Google DeepMind",
      "release_date": "2026-08-13T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Google DeepMind",
        "API access": "Yes",
        "Max output": "64K tokens",
        "Text input": "Yes",
        "Audio input": "Yes",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "gemini-3.7-flash",
        "Audio output": "No",
        "Computer use": "Yes (Preview)",
        "DeepSWE v1.1": "65.3%",
        "Image output": "No",
        "Release date": "2026-08-13",
        "Video output": "No",
        "Context window": "1M tokens",
        "Product access": "Google AI Studio + Gemini API + Gemini Enterprise + Antigravity",
        "Knowledge cutoff": "Mar 2026 (some domains Jan 2025)",
        "Cached input / 1M": "$0.075 introductory through Dec 31, 2026",
        "Input / 1M tokens": "$0.75 introductory through Dec 31, 2026",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$3.75 introductory through Dec 31, 2026",
        "Reasoning / effort": "low / medium / high; medium default",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes",
        "MCP / tool-use benchmark": "FrontierCode 1.1 43.6%"
      },
      "sources": [
        {
          "url": "https://ai.google.dev/gemini-api/docs/models/gemini-3.7-flash",
          "title": "Gemini 3.7 Flash model docs"
        },
        {
          "url": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
          "title": "Gemini 3.7 Flash model card"
        },
        {
          "url": "https://ai.google.dev/gemini-api/docs/pricing",
          "title": "Gemini API pricing"
        }
      ],
      "notes": "Current introductory pricing expires December 31, 2026; standard $1.50/$7.50 begins January 1, 2027.",
      "verified_at": "2026-09-06T16:49:36.933Z"
    },
    {
      "slug": "gemini-3-8-flash",
      "name": "Gemini 3.8 Flash",
      "developer": "Google DeepMind",
      "release_date": "2026-09-02T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Google DeepMind",
        "API access": "Yes",
        "Max output": "64K tokens",
        "Text input": "Yes",
        "Audio input": "Yes",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "gemini-3.8-flash",
        "Audio output": "No",
        "Computer use": "Yes (Preview)",
        "HLE-Verified": "54.9%",
        "Image output": "No",
        "Release date": "2026-09-02",
        "Video output": "No",
        "Context window": "1M tokens",
        "Product access": "Gemini app + Gemini Enterprise + Google AI Studio + Gemini API + AI Mode + Antigravity",
        "Knowledge cutoff": "Mar 2026 (some domains Jan 2025)",
        "Cached input / 1M": "$0.075 introductory through Dec 31, 2026",
        "Input / 1M tokens": "$0.75 introductory through Dec 31, 2026",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$3.75 introductory through Dec 31, 2026",
        "Reasoning / effort": "low / medium / high; medium default",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://ai.google.dev/gemini-api/docs/models/gemini-3.8-flash",
          "title": "Gemini 3.8 Flash model docs"
        },
        {
          "url": "https://deepmind.google/models/gemini/flash/",
          "title": "Gemini 3.8 Flash overview"
        },
        {
          "url": "https://ai.google.dev/gemini-api/docs/pricing",
          "title": "Gemini API pricing"
        }
      ],
      "notes": "Current introductory pricing expires December 31, 2026; standard $1.50/$7.50 begins January 1, 2027.",
      "verified_at": "2026-09-06T16:49:36.933Z"
    },
    {
      "slug": "gemini-omni-1-1-flash",
      "name": "Gemini Omni 1.1 Flash",
      "developer": "Google DeepMind",
      "release_date": "2026-08-27T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Google DeepMind",
        "API access": "Yes",
        "Max output": "Video generation with 4K upscaling; scene extension and frame interpolation",
        "Text input": "Yes",
        "Audio input": "Yes",
        "Text output": "Not primary output",
        "Video input": "Yes",
        "API model ID": "gemini-omni-1.1-flash",
        "Audio output": "Yes, generated with video",
        "Computer use": "No",
        "Image output": "No",
        "Release date": "2026-08-27",
        "Video output": "Yes",
        "Context window": "1M-token family architecture; API-specific limit not separately disclosed on launch page",
        "Product access": "Google AI Studio + Gemini Enterprise Agent Platform + Gemini app + Google Flow",
        "Weights / license": "Proprietary",
        "Image / vision input": "Yes",
        "Tool / function calling": "No general function-calling interface documented"
      },
      "sources": [
        {
          "url": "https://blog.google/innovation-and-ai/technology/developers-tools/build-with-gemini-omni-1-1-flash/",
          "title": "Gemini Omni 1.1 Flash announcement"
        },
        {
          "url": "https://deepmind.google/models/model-cards/gemini-omni-flash/",
          "title": "Gemini Omni Flash model card"
        },
        {
          "url": "https://ai.google.dev/gemini-api/docs/models",
          "title": "Gemini API models"
        }
      ],
      "notes": "Production-ready August update adding scene extension, first/last-frame interpolation and high-resolution upscaling. The current pricing page still labels the Omni preview endpoint, so no unverified 1.1-specific token price is copied here.",
      "verified_at": "2026-09-06T16:49:37.841Z"
    },
    {
      "slug": "gemini-omni-flash",
      "name": "Gemini Omni Flash",
      "developer": "Google DeepMind",
      "release_date": "2026-05-19T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Google DeepMind",
        "API access": "Preview from Jun 30",
        "Max output": "3-10s video, 720p, 24 FPS",
        "Text input": "Yes",
        "Audio input": "Yes (model family/card; preview API docs vary)",
        "Text output": "Not primary output",
        "Video input": "Yes",
        "API model ID": "gemini-omni-flash-preview (API preview Jun 30)",
        "Audio output": "Yes, generated with video",
        "Computer use": "No",
        "Image output": "No",
        "Release date": "2026-05-19",
        "Video output": "Yes",
        "Context window": "1M tokens",
        "Product access": "Gemini app + YouTube + Google Flow; paid Gemini API preview",
        "Input / 1M tokens": "$1.50 text/image/video/audio",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$9 text / $17.50 video (~$0.10/sec at 720p)",
        "Image / vision input": "Yes",
        "Tool / function calling": "No general function-calling interface documented"
      },
      "sources": [
        {
          "url": "https://ai.google.dev/gemini-api/docs/models/gemini-omni-flash",
          "title": "Gemini Omni Flash API model docs"
        },
        {
          "url": "https://deepmind.google/models/model-cards/gemini-omni-flash/",
          "title": "Gemini Omni Flash model card"
        },
        {
          "url": "https://ai.google.dev/gemini-api/docs/pricing",
          "title": "Gemini API pricing"
        }
      ],
      "notes": "Gemini Omni Flash was introduced at I/O in May; the developer API preview followed June 30.",
      "verified_at": "2026-09-06T16:49:37.841Z"
    },
    {
      "slug": "gemma-4-12b",
      "name": "Gemma 4 12B",
      "developer": "Google DeepMind",
      "release_date": "2026-06-03T00:00:00.000Z",
      "access": "open_source",
      "comparison_data": {
        "Developer": "Google DeepMind",
        "API access": "Downloadable weights; self-hosted runtimes",
        "Text input": "Yes",
        "Audio input": "Yes",
        "Text output": "Yes",
        "Video input": "Yes — processed as frame sequences",
        "API model ID": "google/gemma-4-12B-it",
        "Audio output": "No",
        "Computer use": "Via external agent/tool harnesses; no native computer-use API",
        "Image output": "No",
        "Release date": "2026-06-03",
        "Video output": "No",
        "Context window": "256K",
        "Product access": "Kaggle + Hugging Face + self-hosted runtimes",
        "Weights / license": "Open source — Apache 2.0",
        "Reasoning / effort": "Built-in thinking mode",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://ai.google.dev/gemma/docs/core/model_card_4",
          "title": "Gemma 4 model card"
        },
        {
          "url": "https://ai.google.dev/gemma/docs/releases",
          "title": "Gemma release history"
        },
        {
          "url": "https://huggingface.co/collections/google/gemma-4",
          "title": "Gemma 4 model collection"
        }
      ],
      "notes": "11.95B encoder-free unified multimodal model; audio and image streams project directly into the decoder.",
      "verified_at": "2026-09-07T08:16:02.208Z"
    },
    {
      "slug": "gemma-4-26b-a4b",
      "name": "Gemma 4 26B-A4B",
      "developer": "Google DeepMind",
      "release_date": "2026-03-31T00:00:00.000Z",
      "access": "open_source",
      "comparison_data": {
        "Developer": "Google DeepMind",
        "API access": "Gemini API + downloadable weights",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "Yes — processed as frame sequences",
        "API model ID": "gemma-4-26b-a4b-it (Gemini API); google/gemma-4-26B-A4B-it (weights)",
        "Audio output": "No",
        "Computer use": "Via external agent/tool harnesses; no native computer-use API",
        "Image output": "No",
        "Release date": "2026-03-31",
        "Video output": "No",
        "Context window": "256K",
        "Product access": "Google AI Studio / Gemini API + Kaggle/Hugging Face",
        "Weights / license": "Open source — Apache 2.0",
        "Reasoning / effort": "Built-in thinking mode",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://ai.google.dev/gemma/docs/core/model_card_4",
          "title": "Gemma 4 model card"
        },
        {
          "url": "https://ai.google.dev/gemma/docs/releases",
          "title": "Gemma release history"
        },
        {
          "url": "https://huggingface.co/collections/google/gemma-4",
          "title": "Gemma 4 model collection"
        }
      ],
      "notes": "25.2B-total / 3.8B-active MoE with 128 routed experts plus one shared expert; eight routed experts active.",
      "verified_at": "2026-09-07T08:16:02.208Z"
    },
    {
      "slug": "gemma-4-31b",
      "name": "Gemma 4 31B",
      "developer": "Google DeepMind",
      "release_date": "2026-03-31T00:00:00.000Z",
      "access": "open_source",
      "comparison_data": {
        "AIME": "89.2% (AIME 2026, IT Thinking)",
        "MMLU-Pro": "MMMLU 85.2% (IT Thinking)",
        "Developer": "Google DeepMind",
        "API access": "Gemini API + downloadable weights",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "Yes — processed as frame sequences",
        "API model ID": "gemma-4-31b-it (Gemini API); google/gemma-4-31B-it (weights)",
        "Audio output": "No",
        "Computer use": "Via external agent/tool harnesses; no native computer-use API",
        "GPQA Diamond": "84.3% (IT Thinking)",
        "Image output": "No",
        "Release date": "2026-03-31",
        "Video output": "No",
        "LiveCodeBench": "80.0% (v6, IT Thinking)",
        "Context window": "256K",
        "Product access": "Google AI Studio / Gemini API + Kaggle/Hugging Face",
        "Weights / license": "Open source — Apache 2.0",
        "Reasoning / effort": "Built-in thinking mode",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes",
        "MCP / tool-use benchmark": "τ²-bench Retail 86.4% (IT Thinking)"
      },
      "sources": [
        {
          "url": "https://ai.google.dev/gemma/docs/core/model_card_4",
          "title": "Gemma 4 model card"
        },
        {
          "url": "https://ai.google.dev/gemma/docs/releases",
          "title": "Gemma release history"
        },
        {
          "url": "https://huggingface.co/collections/google/gemma-4",
          "title": "Gemma 4 model collection"
        }
      ],
      "notes": "30.7B dense multimodal model with a ~550M vision encoder.",
      "verified_at": "2026-09-07T08:16:02.208Z"
    },
    {
      "slug": "gemma-4-e2b",
      "name": "Gemma 4 E2B",
      "developer": "Google DeepMind",
      "release_date": "2026-03-31T00:00:00.000Z",
      "access": "open_source",
      "comparison_data": {
        "Developer": "Google DeepMind",
        "API access": "Downloadable weights; self-hosted runtimes",
        "Text input": "Yes",
        "Audio input": "Yes",
        "Text output": "Yes",
        "Video input": "Yes — processed as frame sequences",
        "API model ID": "google/gemma-4-E2B-it",
        "Audio output": "No",
        "Computer use": "Via external agent/tool harnesses; no native computer-use API",
        "Image output": "No",
        "Release date": "2026-03-31",
        "Video output": "No",
        "Context window": "128K",
        "Product access": "Kaggle + Hugging Face + self-hosted runtimes",
        "Weights / license": "Open source — Apache 2.0",
        "Reasoning / effort": "Built-in thinking mode",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://ai.google.dev/gemma/docs/core/model_card_4",
          "title": "Gemma 4 model card"
        },
        {
          "url": "https://ai.google.dev/gemma/docs/releases",
          "title": "Gemma release history"
        },
        {
          "url": "https://huggingface.co/collections/google/gemma-4",
          "title": "Gemma 4 model collection"
        }
      ],
      "notes": "2.3B effective / 5.1B including per-layer embeddings; designed for mobile deployment.",
      "verified_at": "2026-09-07T08:16:02.208Z"
    },
    {
      "slug": "gemma-4-e4b",
      "name": "Gemma 4 E4B",
      "developer": "Google DeepMind",
      "release_date": "2026-03-31T00:00:00.000Z",
      "access": "open_source",
      "comparison_data": {
        "Developer": "Google DeepMind",
        "API access": "Downloadable weights; self-hosted runtimes",
        "Text input": "Yes",
        "Audio input": "Yes",
        "Text output": "Yes",
        "Video input": "Yes — processed as frame sequences",
        "API model ID": "google/gemma-4-E4B-it",
        "Audio output": "No",
        "Computer use": "Via external agent/tool harnesses; no native computer-use API",
        "Image output": "No",
        "Release date": "2026-03-31",
        "Video output": "No",
        "Context window": "128K",
        "Product access": "Kaggle + Hugging Face + self-hosted runtimes",
        "Weights / license": "Open source — Apache 2.0",
        "Reasoning / effort": "Built-in thinking mode",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://ai.google.dev/gemma/docs/core/model_card_4",
          "title": "Gemma 4 model card"
        },
        {
          "url": "https://ai.google.dev/gemma/docs/releases",
          "title": "Gemma release history"
        },
        {
          "url": "https://huggingface.co/collections/google/gemma-4",
          "title": "Gemma 4 model collection"
        }
      ],
      "notes": "4.5B effective / 8B including per-layer embeddings; designed for mobile and laptop deployment.",
      "verified_at": "2026-09-07T08:16:02.208Z"
    },
    {
      "slug": "glm-4-7-flash",
      "name": "GLM-4.7-Flash",
      "developer": "Z.ai",
      "release_date": "2026-01-19T00:00:00.000Z",
      "access": "open_source",
      "comparison_data": {
        "Developer": "Z.ai",
        "API access": "Yes",
        "Max output": "128K tokens",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "glm-4.7-flash",
        "Audio output": "No",
        "Computer use": "Via agent integrations; no native computer-use API documented",
        "Image output": "No",
        "Release date": "2026-01-19",
        "Video output": "No",
        "Context window": "200K tokens",
        "Product access": "Z.ai API + open weights",
        "Cache write / 1M": "Free",
        "Cached input / 1M": "Free",
        "Input / 1M tokens": "Free",
        "Weights / license": "Open source — MIT",
        "Output / 1M tokens": "Free",
        "Reasoning / effort": "Multiple thinking modes; preserved thinking for multi-turn agents",
        "Image / vision input": "No",
        "Tool / function calling": "Yes; function calling and agent workflows"
      },
      "sources": [
        {
          "url": "https://huggingface.co/zai-org/GLM-4.7-Flash",
          "title": "GLM-4.7-Flash official model card"
        },
        {
          "url": "https://docs.z.ai/guides/llm/glm-4.7",
          "title": "Z.ai GLM-4.7 docs"
        },
        {
          "url": "https://docs.z.ai/guides/overview/pricing",
          "title": "Z.ai pricing"
        }
      ],
      "notes": "30B-total / 3B-active lightweight MoE. This corrects the older catalog classification from proprietary to MIT open source.",
      "verified_at": "2026-09-07T04:37:00.569Z"
    },
    {
      "slug": "glm-5",
      "name": "GLM-5",
      "developer": "Z.ai",
      "release_date": "2026-02-12T00:00:00.000Z",
      "access": "open_source",
      "comparison_data": {
        "Developer": "Z.ai",
        "API access": "Yes",
        "Max output": "128K tokens",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "glm-5",
        "Audio output": "No",
        "Computer use": "Via agent harnesses; no native computer-use API documented",
        "Image output": "No",
        "Release date": "2026-02-12",
        "Video output": "No",
        "Context window": "200K tokens",
        "Product access": "Z.ai API + GLM Coding Plan + open weights",
        "Cached input / 1M": "$0.20",
        "Input / 1M tokens": "$1.00",
        "Weights / license": "Open source — MIT",
        "Output / 1M tokens": "$3.20",
        "Reasoning / effort": "Thinking model",
        "Image / vision input": "No",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://huggingface.co/zai-org/GLM-5",
          "title": "GLM-5 official model card"
        },
        {
          "url": "https://docs.z.ai/guides/llm/glm-5",
          "title": "Z.ai GLM-5 docs"
        },
        {
          "url": "https://docs.z.ai/guides/overview/pricing",
          "title": "Z.ai pricing"
        }
      ],
      "notes": "744B-total / 40B-active MoE for agentic engineering and long-horizon tasks, using DeepSeek Sparse Attention.",
      "verified_at": "2026-09-07T04:37:00.569Z"
    },
    {
      "slug": "glm-5-1",
      "name": "GLM-5.1",
      "developer": "Z.ai",
      "release_date": "2026-04-07T00:00:00.000Z",
      "access": "open_source",
      "comparison_data": {
        "Developer": "Z.ai",
        "API access": "Yes",
        "Max output": "128K tokens",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "glm-5.1",
        "Audio output": "No",
        "Computer use": "Via coding/agent harnesses; no native computer-use API documented",
        "Image output": "No",
        "Release date": "2026-04-07",
        "Video output": "No",
        "Context window": "200K tokens",
        "Product access": "Z.ai API + GLM Coding Plan + open weights",
        "Cached input / 1M": "$0.26",
        "Input / 1M tokens": "$1.40",
        "Weights / license": "Open source — MIT",
        "Output / 1M tokens": "$4.40",
        "Reasoning / effort": "Deep thinking; long-horizon autonomous execution",
        "Image / vision input": "No",
        "Tool / function calling": "Yes; streaming tool calls supported"
      },
      "sources": [
        {
          "url": "https://huggingface.co/zai-org/GLM-5.1",
          "title": "GLM-5.1 official model card"
        },
        {
          "url": "https://docs.z.ai/guides/llm/glm-5.1",
          "title": "Z.ai GLM-5.1 docs"
        },
        {
          "url": "https://docs.z.ai/guides/overview/pricing",
          "title": "Z.ai pricing"
        }
      ],
      "notes": "Post-trained GLM-5-family flagship built for sustained autonomous work over hundreds of rounds and thousands of tool calls. This corrects the older proprietary catalog classification to MIT open source.",
      "verified_at": "2026-09-07T04:37:00.569Z"
    },
    {
      "slug": "glm-5-2",
      "name": "GLM-5.2",
      "developer": "Z.ai",
      "release_date": "2026-06-24T00:00:00.000Z",
      "access": "open_source",
      "comparison_data": {
        "Developer": "Z.ai",
        "API access": "Yes",
        "Max output": "128K tokens",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "glm-5.2",
        "Audio output": "No",
        "Computer use": "Via agent harnesses; no native computer-use API documented",
        "Image output": "No",
        "Release date": "2026-06-24",
        "Video output": "No",
        "Context window": "1M tokens",
        "Product access": "Z.ai API + open weights",
        "Cached input / 1M": "$0.26",
        "Input / 1M tokens": "$1.40",
        "Weights / license": "Open source — MIT",
        "Output / 1M tokens": "$4.40",
        "Reasoning / effort": "Multiple thinking modes; low / high / max supported",
        "Image / vision input": "No",
        "Tool / function calling": "Yes; function calling, MCP and structured output"
      },
      "sources": [
        {
          "url": "https://docs.z.ai/guides/llm/glm-5.2",
          "title": "GLM-5.2 developer docs"
        },
        {
          "url": "https://huggingface.co/zai-org/GLM-5.2",
          "title": "GLM-5.2 official weights"
        },
        {
          "url": "https://docs.z.ai/guides/overview/pricing",
          "title": "Z.ai pricing"
        },
        {
          "url": "https://autoclaw.z.ai/blog/model/glm-5.3-flash/",
          "title": "GLM-5.3-Flash comparison"
        }
      ],
      "notes": "753B-parameter open-source flagship with a solid 1M context window. Pricing is Z.ai's current API rate.",
      "verified_at": "2026-09-07T04:00:13.944Z"
    },
    {
      "slug": "glm-5-3",
      "name": "GLM-5.3",
      "developer": "Z.ai",
      "release_date": "2026-08-28T00:00:00.000Z",
      "access": "open_weights",
      "comparison_data": {
        "Developer": "Z.ai",
        "API access": "Yes",
        "Max output": "128K tokens",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "glm-5.3; ZHIPU/GLM-5.3 on Alibaba Model Studio",
        "Audio output": "No",
        "Computer use": "Via agent harnesses; no native computer-use API documented",
        "Image output": "No",
        "Release date": "2026-08-28",
        "Video output": "No",
        "Context window": "1M tokens",
        "Product access": "Z.ai API + open weights + third-party hosting",
        "Cached input / 1M": "$0.26 implicit cache via Alibaba Model Studio Singapore",
        "Input / 1M tokens": "$1.40 hosted via Alibaba Model Studio Singapore",
        "Weights / license": "Open weights — GLM-5.3 License",
        "Output / 1M tokens": "$4.40 hosted via Alibaba Model Studio Singapore",
        "Reasoning / effort": "low / high / max; max default",
        "Image / vision input": "No",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://huggingface.co/zai-org/GLM-5.3",
          "title": "GLM-5.3 official model card"
        },
        {
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "title": "GLM-5.3 on Alibaba Model Studio"
        }
      ],
      "notes": "Text-only GLM-5.2-base derivative whose gains come primarily from post-training. Hosted price fields are explicitly the Alibaba Model Studio Singapore rates, not a claimed universal Z.ai API price.",
      "verified_at": "2026-09-07T03:17:35.772Z"
    },
    {
      "slug": "glm-5-3-flash",
      "name": "GLM-5.3-Flash",
      "developer": "Z.ai",
      "release_date": "2026-08-26T00:00:00.000Z",
      "access": "open_source",
      "comparison_data": {
        "Developer": "Z.ai",
        "API access": "Yes",
        "Max output": "128K tokens",
        "Text input": "Yes",
        "Audio input": "No native audio input documented",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "glm-5.3-flash",
        "Audio output": "No",
        "Computer use": "Yes; visual feedback and GUI-agent workflows",
        "DeepSWE v1.1": "63.4%",
        "GDPval-AA v2": "1,773 Elo",
        "Image output": "No",
        "Release date": "2026-08-26",
        "Video output": "No",
        "Context window": "1M tokens",
        "Product access": "Z.ai API + AutoClaw + open weights",
        "Terminal-Bench": "84.3% (Terminal-Bench 2.1; Z.ai report)",
        "Cached input / 1M": "$0.03 list ($0.015 promo through Sep 9, 2026)",
        "Input / 1M tokens": "$0.15 list ($0.075 promo through Sep 9, 2026)",
        "Weights / license": "Open source — MIT",
        "Output / 1M tokens": "$0.50 list ($0.25 promo through Sep 9, 2026)",
        "Reasoning / effort": "Reasoning enabled; low / high / max effort",
        "Terminal-Bench 2.1": "84.3%",
        "Humanity's Last Exam": "55.3% with tools",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes; tools, structured output and prompt caching",
        "MCP / tool-use benchmark": "Toolathlon Verified 78.4%; AutomationBench 48.8%; Agents' Last Exam 26.3%"
      },
      "sources": [
        {
          "url": "https://huggingface.co/zai-org/GLM-5.3-Flash",
          "title": "GLM-5.3-Flash official model card"
        },
        {
          "url": "https://z.ai/blog/glm-5.3-flash",
          "title": "GLM-5.3-Flash launch"
        },
        {
          "url": "https://docs.z.ai/guides/overview/pricing",
          "title": "Z.ai pricing"
        }
      ],
      "notes": "320B-total / 18B-active native multimodal model with hybrid sparse + linear attention. Catalog records list pricing and separately notes the temporary 50% launch promotion.",
      "verified_at": "2026-09-07T04:37:00.569Z"
    },
    {
      "slug": "glm-5-turbo",
      "name": "GLM-5-Turbo",
      "developer": "Z.ai",
      "release_date": "2026-03-15T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Z.ai",
        "API access": "Yes",
        "Max output": "128K tokens",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "glm-5-turbo",
        "Audio output": "No",
        "Computer use": "Optimized for OpenClaw-style agent execution",
        "Image output": "No",
        "Release date": "2026-03-15",
        "Video output": "No",
        "Context window": "200K tokens",
        "Product access": "Z.ai API + GLM Coding Plan",
        "Cached input / 1M": "$0.24",
        "Input / 1M tokens": "$1.20",
        "Weights / license": "Proprietary hosted model",
        "Output / 1M tokens": "$4.00",
        "Reasoning / effort": "Multiple thinking modes",
        "Image / vision input": "No",
        "Tool / function calling": "Yes; function calling, MCP, structured output and context caching"
      },
      "sources": [
        {
          "url": "https://docs.z.ai/guides/llm/glm-5-turbo",
          "title": "Z.ai GLM-5-Turbo docs"
        },
        {
          "url": "https://docs.z.ai/guides/overview/pricing",
          "title": "Z.ai pricing"
        },
        {
          "url": "https://docs.z.ai/release-notes/new-released",
          "title": "Z.ai release notes"
        }
      ],
      "notes": "Hosted GLM-5 variant optimized from training onward for OpenClaw tasks, long chains, tool invocation, scheduled/persistent work and instruction decomposition.",
      "verified_at": "2026-09-07T04:37:00.569Z"
    },
    {
      "slug": "glm-5v-turbo",
      "name": "GLM-5V-Turbo",
      "developer": "Z.ai",
      "release_date": "2026-04-01T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Z.ai",
        "API access": "Yes",
        "Max output": "128K tokens",
        "Text input": "Yes",
        "Audio input": "No native audio input documented",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "glm-5v-turbo",
        "Audio output": "No",
        "Computer use": "Yes; optimized for GUI/vision-agent workflows",
        "Image output": "No",
        "Release date": "2026-04-01",
        "Video output": "No",
        "Context window": "200K tokens",
        "Product access": "Z.ai API + agent integrations",
        "Cached input / 1M": "$0.24",
        "Input / 1M tokens": "$1.20",
        "Weights / license": "Proprietary hosted model",
        "Output / 1M tokens": "$4.00",
        "Reasoning / effort": "Multiple thinking modes",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://docs.z.ai/guides/vlm/glm-5v-turbo",
          "title": "Z.ai GLM-5V-Turbo docs"
        },
        {
          "url": "https://docs.z.ai/guides/overview/pricing",
          "title": "Z.ai pricing"
        },
        {
          "url": "https://docs.z.ai/release-notes/new-released",
          "title": "Z.ai release notes"
        }
      ],
      "notes": "Z.ai's first multimodal coding foundation model, accepting video, images, text and files and optimized for end-to-end visual agent workflows.",
      "verified_at": "2026-09-07T04:37:00.569Z"
    },
    {
      "slug": "gpt-5-3-codex",
      "name": "GPT-5.3-Codex",
      "developer": "OpenAI",
      "release_date": "2026-02-05T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "OpenAI",
        "API access": "Yes — Chat Completions, Responses and Batch",
        "Max output": "128K",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "gpt-5.3-codex",
        "Audio output": "No",
        "Computer use": "Yes — optimized for Codex and agentic coding environments",
        "Image output": "No native output",
        "Release date": "2026-02-05",
        "Video output": "No",
        "SWE-bench Pro": "56.8% (OpenAI comparison eval)",
        "Context window": "400K",
        "Product access": "Codex + OpenAI API",
        "Knowledge cutoff": "Aug 31, 2025",
        "Cached input / 1M": "$0.175",
        "Input / 1M tokens": "$1.75",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$14",
        "Reasoning / effort": "low / medium / high / xhigh",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://developers.openai.com/api/docs/models/gpt-5.3-codex",
          "title": "GPT-5.3-Codex API model page"
        },
        {
          "url": "https://openai.com/index/introducing-gpt-5-3-codex/",
          "title": "Introducing GPT-5.3-Codex"
        },
        {
          "url": "https://openai.com/index/introducing-gpt-5-4/",
          "title": "Introducing GPT-5.4"
        }
      ],
      "notes": "Agentic coding model. The OSWorld-Verified value stored by this migration uses OpenAI's revised 74.0% result, which supersedes the original 64.7% after an image-resolution-preserving API parameter was introduced.",
      "verified_at": "2026-09-07T09:26:57.038Z"
    },
    {
      "slug": "gpt-5-3-codex-spark",
      "name": "GPT-5.3-Codex-Spark",
      "developer": "OpenAI",
      "release_date": "2026-02-12T00:00:00.000Z",
      "access": "restricted",
      "comparison_data": {
        "Developer": "OpenAI",
        "API access": "Limited design-partner preview",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "gpt-5.3-codex-spark (research preview)",
        "Audio output": "No",
        "Computer use": "Codex app / CLI / VS Code tool environment",
        "Image output": "No",
        "Release date": "2026-02-12",
        "Video output": "No",
        "Context window": "128K",
        "Product access": "ChatGPT Pro via Codex app, CLI and VS Code extension",
        "Weights / license": "Restricted proprietary research preview",
        "Reasoning / effort": "Optimized for lightweight real-time coding",
        "Image / vision input": "No",
        "Tool / function calling": "Yes — through Codex environment"
      },
      "sources": [
        {
          "url": "https://openai.com/index/introducing-gpt-5-3-codex-spark/",
          "title": "Introducing GPT-5.3-Codex-Spark"
        }
      ],
      "notes": "Research preview served on Cerebras and designed for >1,000 output tokens/s. OpenAI publishes qualitative SWE-Bench Pro and Terminal-Bench 2.0 comparisons but not machine-readable numeric scores, so no benchmark values are inferred here.",
      "verified_at": "2026-09-07T09:26:57.038Z"
    },
    {
      "slug": "gpt-5-3-instant",
      "name": "GPT-5.3 Instant",
      "developer": "OpenAI",
      "release_date": "2026-03-03T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "OpenAI",
        "API access": "Deprecated API alias — gpt-5.3-chat-latest",
        "Max output": "16K",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "gpt-5.3-chat-latest",
        "Audio output": "No",
        "Computer use": "No native computer-use model support documented",
        "Image output": "No native output",
        "Release date": "2026-03-03",
        "Video output": "No",
        "Context window": "128K",
        "Product access": "Legacy ChatGPT model; replaced by GPT-5.5 Instant",
        "Knowledge cutoff": "Aug 31, 2025",
        "Cached input / 1M": "$0.175",
        "Input / 1M tokens": "$1.75",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$14",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://developers.openai.com/api/docs/models/gpt-5.3-chat-latest",
          "title": "GPT-5.3 Chat API model page"
        },
        {
          "url": "https://openai.com/index/gpt-5-3-instant/",
          "title": "GPT-5.3 Instant launch"
        }
      ],
      "notes": "Former ChatGPT default Instant model. OpenAI has deprecated the API alias and recommends newer models; no comparison-safe public benchmark table was published with the product update.",
      "verified_at": "2026-09-07T09:26:57.038Z"
    },
    {
      "slug": "gpt-5-4",
      "name": "GPT-5.4",
      "developer": "OpenAI",
      "release_date": "2026-03-05T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "OSWorld": "75.0% (OSWorld-Verified)",
        "Developer": "OpenAI",
        "API access": "Yes — OpenAI API",
        "Max output": "128K",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "gpt-5.4",
        "Audio output": "No",
        "Computer use": "Yes — native Responses API computer use",
        "Image output": "No native output; image-generation tool supported",
        "Release date": "2026-03-05",
        "Video output": "No",
        "SWE-bench Pro": "57.7%",
        "Context window": "1.05M",
        "Product access": "ChatGPT + Codex + OpenAI API",
        "Knowledge cutoff": "Aug 31, 2025",
        "Cached input / 1M": "$0.25",
        "Input / 1M tokens": "$2.50",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$15",
        "Reasoning / effort": "none (default) / low / medium / high / xhigh",
        "Terminal-Bench 2.1": "75.1% (Terminal-Bench 2.0 in launch eval)",
        "Image / vision input": "Yes",
        "Long-context surcharge": ">272K input: 2× input and 1.5× output for the full session",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://developers.openai.com/api/docs/models/gpt-5.4",
          "title": "GPT-5.4 API model page"
        },
        {
          "url": "https://openai.com/index/introducing-gpt-5-4/",
          "title": "Introducing GPT-5.4"
        },
        {
          "url": "https://openai.com/index/introducing-gpt-5-5/",
          "title": "Introducing GPT-5.5"
        }
      ],
      "notes": "General-purpose reasoning model with native computer use and tool search. Benchmark rows use OpenAI's latest published reruns where the GPT-5.5 launch updated prior GPT-5.4 values.",
      "verified_at": "2026-09-07T09:26:57.038Z"
    },
    {
      "slug": "gpt-5-4-mini",
      "name": "GPT-5.4 mini",
      "developer": "OpenAI",
      "release_date": "2026-03-17T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "OpenAI",
        "API access": "Yes",
        "Max output": "128K tokens",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "gpt-5.4-mini",
        "Audio output": "No",
        "Computer use": "Yes",
        "Image output": "No native image output; image-generation tool supported",
        "Release date": "2026-03-17",
        "Video output": "No",
        "Context window": "400K tokens",
        "Product access": "OpenAI API + Codex + ChatGPT",
        "Knowledge cutoff": "2025-08-31",
        "Cached input / 1M": "$0.075",
        "Input / 1M tokens": "$0.75",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$4.50",
        "Reasoning / effort": "none (default) / low / medium / high / xhigh",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
          "title": "Introducing GPT-5.4 mini and nano"
        },
        {
          "url": "https://developers.openai.com/api/docs/models/gpt-5.4-mini",
          "title": "GPT-5.4 mini API model page"
        }
      ],
      "notes": "OpenAI describes GPT-5.4 mini as optimized for coding, computer use, high-volume workloads, and subagents.",
      "verified_at": "2026-09-06T15:24:25.867Z"
    },
    {
      "slug": "gpt-5-4-nano",
      "name": "GPT-5.4 nano",
      "developer": "OpenAI",
      "release_date": "2026-03-17T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "OpenAI",
        "API access": "Yes",
        "Max output": "128K tokens",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "gpt-5.4-nano",
        "Audio output": "No",
        "Computer use": "No",
        "Image output": "No native image output; image-generation tool supported",
        "Release date": "2026-03-17",
        "Video output": "No",
        "Context window": "400K tokens",
        "Product access": "OpenAI API only",
        "Knowledge cutoff": "2025-08-31",
        "Cached input / 1M": "$0.02",
        "Input / 1M tokens": "$0.20",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$1.25",
        "Reasoning / effort": "none (default) / low / medium / high / xhigh",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
          "title": "Introducing GPT-5.4 mini and nano"
        },
        {
          "url": "https://developers.openai.com/api/docs/models/gpt-5.4-nano",
          "title": "GPT-5.4 nano API model page"
        }
      ],
      "notes": "OpenAI positions GPT-5.4 nano for classification, data extraction, ranking, and simpler coding subagents. Its API model page explicitly lists computer use as unsupported.",
      "verified_at": "2026-09-06T15:24:25.867Z"
    },
    {
      "slug": "gpt-5-4-pro",
      "name": "GPT-5.4 Pro",
      "developer": "OpenAI",
      "release_date": "2026-03-05T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "OpenAI",
        "API access": "Responses API + Batch API",
        "Max output": "128K",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "gpt-5.4-pro",
        "Audio output": "No",
        "Computer use": "Yes",
        "Image output": "No native output; image-generation tool supported",
        "Release date": "2026-03-05",
        "Video output": "No",
        "Context window": "1.05M",
        "Product access": "ChatGPT Pro + OpenAI API",
        "Knowledge cutoff": "Aug 31, 2025",
        "Cached input / 1M": "No cached-input discount",
        "Input / 1M tokens": "$30",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$180",
        "Reasoning / effort": "medium (default) / high / xhigh",
        "Image / vision input": "Yes",
        "Long-context surcharge": ">272K input: 2× input and 1.5× output for the full session",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://developers.openai.com/api/docs/models/gpt-5.4-pro",
          "title": "GPT-5.4 Pro API model page"
        },
        {
          "url": "https://openai.com/index/introducing-gpt-5-4/",
          "title": "Introducing GPT-5.4"
        },
        {
          "url": "https://openai.com/index/introducing-gpt-5-5/",
          "title": "Introducing GPT-5.5"
        }
      ],
      "notes": "Higher-compute GPT-5.4 variant. Responses API is the supported interactive API surface; no cached-input discount is offered.",
      "verified_at": "2026-09-07T09:26:57.038Z"
    },
    {
      "slug": "gpt-5-5",
      "name": "GPT-5.5",
      "developer": "OpenAI",
      "release_date": "2026-04-23T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "OSWorld": "78.7% (OSWorld-Verified)",
        "Developer": "OpenAI",
        "API access": "Yes",
        "BrowseComp": "84.4%",
        "Max output": "128K tokens",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "gpt-5.5",
        "Audio output": "No",
        "Computer use": "Yes",
        "FrontierMath": "51.7% Tiers 1–3 / 35.4% Tier 4",
        "GPQA Diamond": "93.6%",
        "Image output": "No native image output; image-generation tool supported",
        "Release date": "2026-04-23",
        "Video output": "No",
        "SWE-bench Pro": "58.6%",
        "Context window": "1.05M tokens",
        "Product access": "ChatGPT + Codex + OpenAI API",
        "Knowledge cutoff": "2025-12-01",
        "Cached input / 1M": "$0.50",
        "Input / 1M tokens": "$5.00",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$30.00",
        "Reasoning / effort": "none / low / medium / high / xhigh",
        "Image / vision input": "Yes",
        "Batch / flex discount": "50% off standard token rates",
        "Long-context surcharge": ">272K input: 2× input and 1.5× output for the full session",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://openai.com/index/introducing-gpt-5-5/",
          "title": "Introducing GPT-5.5"
        },
        {
          "url": "https://developers.openai.com/api/docs/models/gpt-5.5",
          "title": "GPT-5.5 API model page"
        }
      ],
      "notes": "GPT-5.5 launched in ChatGPT and Codex on April 23; API availability followed on April 24, 2026.",
      "verified_at": "2026-09-06T15:24:25.867Z"
    },
    {
      "slug": "gpt-5-5-instant",
      "name": "GPT-5.5 Instant",
      "developer": "OpenAI",
      "release_date": "2026-05-05T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "OpenAI",
        "API access": "Yes — chat-latest rolling alias",
        "Max output": "128K",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "chat-latest (GPT-5.5 Instant rolling alias)",
        "Audio output": "No",
        "Computer use": "No",
        "Image output": "No native output; image-generation tool supported",
        "Release date": "2026-05-05",
        "Video output": "No",
        "Context window": "400K",
        "Product access": "Default Instant model in ChatGPT + OpenAI API",
        "Knowledge cutoff": "Aug 31, 2025",
        "Cached input / 1M": "$0.50",
        "Input / 1M tokens": "$5",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$30",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://developers.openai.com/api/docs/models/chat-latest",
          "title": "Chat Latest API model page"
        },
        {
          "url": "https://openai.com/index/gpt-5-5-instant/",
          "title": "GPT-5.5 Instant launch"
        },
        {
          "url": "https://openai.com/index/gpt-5-5-instant-system-card/",
          "title": "GPT-5.5 Instant system card"
        }
      ],
      "notes": "GPT-5.5 Instant replaced GPT-5.3 Instant as ChatGPT's default Instant model. The API uses the rolling chat-latest alias, whose underlying snapshot can change; no stable comparison-safe benchmark suite is attached to that alias.",
      "verified_at": "2026-09-07T09:26:57.038Z"
    },
    {
      "slug": "gpt-5-5-pro",
      "name": "GPT-5.5 Pro",
      "developer": "OpenAI",
      "release_date": "2026-04-23T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "OpenAI",
        "API access": "Responses API + Batch API",
        "Max output": "128K",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "gpt-5.5-pro",
        "Audio output": "No",
        "Computer use": "No",
        "Image output": "No native output; image-generation tool supported",
        "Release date": "2026-04-23",
        "Video output": "No",
        "Context window": "1.05M",
        "Product access": "ChatGPT Pro / Business / Enterprise + OpenAI API",
        "Knowledge cutoff": "Dec 1, 2025",
        "Cached input / 1M": "No cached-input discount",
        "Input / 1M tokens": "$30",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$180",
        "Reasoning / effort": "medium / high (default) / xhigh",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://developers.openai.com/api/docs/models/gpt-5.5-pro",
          "title": "GPT-5.5 Pro API model page"
        },
        {
          "url": "https://openai.com/index/introducing-gpt-5-5/",
          "title": "Introducing GPT-5.5"
        }
      ],
      "notes": "Higher-compute GPT-5.5 variant. OpenAI explicitly lists computer use as unsupported and does not offer a cached-input discount.",
      "verified_at": "2026-09-07T09:26:57.038Z"
    },
    {
      "slug": "gpt-5-6-luna",
      "name": "GPT-5.6 Luna",
      "developer": "OpenAI",
      "release_date": "2026-07-09T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "OpenAI",
        "API access": "Yes",
        "Max output": "128K tokens",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "gpt-5.6-luna",
        "Audio output": "No",
        "Computer use": "Yes",
        "FrontierMath": "58.5% Tier 4 v2",
        "GPQA Diamond": "92.3%",
        "Image output": "No native image output; image-generation tool supported",
        "Release date": "2026-07-09",
        "Video output": "No",
        "Context window": "1.05M tokens",
        "Product access": "ChatGPT / ChatGPT Work + Codex + OpenAI API",
        "Cache write / 1M": "$0.25 (1.25× uncached input)",
        "Knowledge cutoff": "2026-02-16",
        "Cached input / 1M": "$0.02",
        "Input / 1M tokens": "$0.20",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$1.20",
        "Reasoning / effort": "none / low / medium / high / xhigh / max",
        "Image / vision input": "Yes",
        "Long-context surcharge": ">272K input: 2× input and 1.5× output for the full request",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://openai.com/index/gpt-5-6/",
          "title": "GPT-5.6 launch"
        },
        {
          "url": "https://developers.openai.com/api/docs/models/gpt-5.6-luna",
          "title": "GPT-5.6 Luna API model page"
        },
        {
          "url": "https://openai.com/index/advancing-the-price-performance-frontier-with-gpt-5-6/",
          "title": "GPT-5.6 current price-performance update"
        }
      ],
      "notes": "Luna is the fastest and most affordable GPT-5.6 tier. OpenAI reduced its API price on July 30, 2026 to $0.20 input / $1.20 output per 1M tokens.",
      "verified_at": "2026-09-06T15:24:25.867Z"
    },
    {
      "slug": "gpt-5-6-sol",
      "name": "GPT-5.6 Sol",
      "developer": "OpenAI",
      "release_date": "2026-07-09T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "OpenAI",
        "API access": "Yes",
        "BrowseComp": "90.4% standard / 92.2% Ultra",
        "Max output": "128K tokens",
        "Text input": "Yes",
        "Audio input": "No",
        "OSWorld 2.0": "62.6%",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "gpt-5.6-sol (gpt-5.6 alias routes to Sol)",
        "Audio output": "No",
        "Computer use": "Yes",
        "FrontierMath": "89.0% Tiers 1–3 v2 / 83.0% Tier 4 v2",
        "GPQA Diamond": "94.6%",
        "Image output": "No native image output; image-generation tool supported",
        "Release date": "2026-07-09",
        "Video output": "No",
        "Context window": "1.05M tokens",
        "Product access": "ChatGPT / ChatGPT Work + Codex + OpenAI API",
        "Cache write / 1M": "$5.00 (1.25× uncached input)",
        "Knowledge cutoff": "2026-02-16",
        "Cached input / 1M": "$0.40",
        "Input / 1M tokens": "$4.00 (current promotional API price)",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$20.00 (current promotional API price)",
        "Reasoning / effort": "none / low / medium / high / xhigh / max",
        "Image / vision input": "Yes",
        "Long-context surcharge": ">272K input: 2× input and 1.5× output for the full request",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://openai.com/index/gpt-5-6/",
          "title": "GPT-5.6 launch"
        },
        {
          "url": "https://developers.openai.com/api/docs/models/gpt-5.6-sol",
          "title": "GPT-5.6 Sol API model page"
        },
        {
          "url": "https://openai.com/index/advancing-the-price-performance-frontier-with-gpt-5-6/",
          "title": "GPT-5.6 current price-performance update"
        }
      ],
      "notes": "GPT-5.6 Sol launched GA on July 9 after a June 26 limited preview. Current $4/$20 API pricing is promotional and OpenAI says it is available at least through November 21, 2026.",
      "verified_at": "2026-09-06T15:24:25.867Z"
    },
    {
      "slug": "gpt-5-6-terra",
      "name": "GPT-5.6 Terra",
      "developer": "OpenAI",
      "release_date": "2026-07-09T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "OpenAI",
        "API access": "Yes",
        "Max output": "128K tokens",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "gpt-5.6-terra",
        "Audio output": "No",
        "Computer use": "Yes",
        "FrontierMath": "68.3% Tier 4 v2",
        "GPQA Diamond": "92.9%",
        "Image output": "No native image output; image-generation tool supported",
        "Release date": "2026-07-09",
        "Video output": "No",
        "Context window": "1.05M tokens",
        "Product access": "ChatGPT / ChatGPT Work + Codex + OpenAI API",
        "Cache write / 1M": "$2.50 (1.25× uncached input)",
        "Knowledge cutoff": "2026-02-16",
        "Cached input / 1M": "$0.20",
        "Input / 1M tokens": "$2.00",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$12.00",
        "Reasoning / effort": "none / low / medium / high / xhigh / max",
        "Image / vision input": "Yes",
        "Long-context surcharge": ">272K input: 2× input and 1.5× output for the full request",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://openai.com/index/gpt-5-6/",
          "title": "GPT-5.6 launch"
        },
        {
          "url": "https://developers.openai.com/api/docs/models/gpt-5.6-terra",
          "title": "GPT-5.6 Terra API model page"
        },
        {
          "url": "https://openai.com/index/advancing-the-price-performance-frontier-with-gpt-5-6/",
          "title": "GPT-5.6 current price-performance update"
        }
      ],
      "notes": "Terra is the balanced GPT-5.6 tier. OpenAI reduced its API price on July 30, 2026 to $2 input / $12 output per 1M tokens.",
      "verified_at": "2026-09-06T15:24:25.867Z"
    },
    {
      "slug": "gpt-6-astra",
      "name": "GPT-6 Astra",
      "developer": "OpenAI",
      "release_date": "2026-09-03T00:00:00.000Z",
      "access": "restricted",
      "comparison_data": {
        "ARC-AGI": "99.9% (ARC-AGI-3)",
        "Developer": "OpenAI",
        "API access": "Limited rollout; documented in OpenAI API",
        "Max output": "128K tokens",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "gpt-6-astra",
        "Audio output": "No",
        "Computer use": "Yes",
        "FrontierMath": "98% (Tier 4 launch eval)",
        "Image output": "No native image output; image-generation tool supported",
        "Release date": "2026-09-03",
        "Video output": "No",
        "Context window": "1.05M tokens",
        "Product access": "Trusted Access enterprises; API and Plus/Pro/Business/Enterprise rollout",
        "Cache write / 1M": "$12.50",
        "Knowledge cutoff": "Apr 30, 2026",
        "Cached input / 1M": "$1",
        "Input / 1M tokens": "$10",
        "Weights / license": "Restricted proprietary rollout",
        "Output / 1M tokens": "$50",
        "Reasoning / effort": "low / medium / high / xhigh / max",
        "Image / vision input": "Yes",
        "Long-context surcharge": ">272K input: 2× input/cache and 1.5× output for the full request",
        "Tool / function calling": "Yes",
        "MCP / tool-use benchmark": "ExploitBench 100% (restricted rollout evaluation)"
      },
      "sources": [
        {
          "url": "https://openai.com/index/gpt-6-astra/",
          "title": "GPT-6 Astra launch"
        },
        {
          "url": "https://developers.openai.com/api/docs/models/gpt-6-astra",
          "title": "GPT-6 Astra API model page"
        }
      ],
      "notes": "Released September 3, 2026 in a staged rollout. Catalog access remains restricted until OpenAI completes general rollout.",
      "verified_at": "2026-09-06T16:49:35.977Z"
    },
    {
      "slug": "grok-4-5",
      "name": "Grok 4.5",
      "developer": "xAI",
      "release_date": "2026-07-16T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "xAI",
        "API access": "Yes",
        "Text input": "Yes",
        "Audio input": "No native audio input documented",
        "Text output": "Yes",
        "Video input": "No native video input documented",
        "API model ID": "grok-4.5",
        "Audio output": "No",
        "Computer use": "Via Grok Build / agent environments",
        "DeepSWE v1.1": "62.0% (DeepSWE 1.0)",
        "Image output": "No",
        "Release date": "2026-07-16",
        "Video output": "No",
        "SWE-bench Pro": "64.7%",
        "Context window": "500K tokens",
        "Product access": "Grok Build + Cursor + GitHub Copilot + xAI API",
        "Cached input / 1M": "$0.30 <=200K / $0.60 >200K",
        "Input / 1M tokens": "$2 <=200K / $4 >200K",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$6 <=200K / $12 >200K",
        "Reasoning / effort": "low / medium / high; high default",
        "Terminal-Bench 2.1": "83.3%",
        "Image / vision input": "Yes",
        "Long-context surcharge": ">=200K prompt tokens use the higher tier",
        "Tool / function calling": "Yes",
        "MCP / tool-use benchmark": "SWE Marathon 29.0%"
      },
      "sources": [
        {
          "url": "https://x.ai/news/grok-4-5",
          "title": "Introducing Grok 4.5"
        },
        {
          "url": "https://docs.x.ai/developers/models/grok-4.5",
          "title": "Grok 4.5 model docs"
        },
        {
          "url": "https://docs.x.ai/developers/pricing",
          "title": "xAI pricing"
        }
      ],
      "notes": "Coding- and agent-focused model. xAI's API release notes show API availability before the July 16 announcement; the catalog release date tracks the public model announcement.",
      "verified_at": "2026-09-07T04:00:13.944Z"
    },
    {
      "slug": "grok-4-6",
      "name": "Grok 4.6",
      "developer": "xAI",
      "release_date": "2026-08-12T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "xAI",
        "API access": "Yes",
        "Max output": "No text output limit",
        "Text input": "Yes",
        "Audio input": "No native audio input documented",
        "CursorBench": "69.9% (CursorBench 3.2)",
        "Text output": "Yes",
        "Video input": "No native video input documented",
        "API model ID": "grok-4.6",
        "Audio output": "No",
        "Computer use": "Via Grok Build / agent environments",
        "DeepSWE v1.1": "65.9%",
        "GDPval-AA v2": "1,753 Elo",
        "Image output": "No",
        "Release date": "2026-08-12",
        "Video output": "No",
        "Context window": "500K tokens",
        "Product access": "Grok Build + Cursor + xAI API + partner gateways",
        "Knowledge cutoff": "Jan 2026",
        "Cached input / 1M": "$0.50 <=200K / $1 >200K",
        "Input / 1M tokens": "$2 <=200K / $4 >200K",
        "Weights / license": "Proprietary",
        "Output / 1M tokens": "$6 <=200K / $12 >200K",
        "Reasoning / effort": "low / medium / high / xhigh; high default",
        "Image / vision input": "Yes",
        "Long-context surcharge": ">=200K prompt tokens use the higher tier",
        "Tool / function calling": "Yes; function calling, web search, X search, code execution",
        "MCP / tool-use benchmark": "FrontierCode 1.1 Extended 61.3%; APEX-Agents 57.5%"
      },
      "sources": [
        {
          "url": "https://x.ai/news/grok-4-6",
          "title": "Introducing Grok 4.6"
        },
        {
          "url": "https://docs.x.ai/developers/models/grok-4.6",
          "title": "Grok 4.6 model docs"
        },
        {
          "url": "https://docs.x.ai/developers/pricing",
          "title": "xAI pricing"
        }
      ],
      "notes": "Current frontier coding/agent model. Fast variant is available at twice the standard token price.",
      "verified_at": "2026-09-07T04:00:13.944Z"
    },
    {
      "slug": "jamba2-3b",
      "name": "Jamba2 3B",
      "developer": "AI21 Labs",
      "release_date": "2026-01-08T00:00:00.000Z",
      "access": "open_source",
      "comparison_data": {
        "Developer": "AI21 Labs",
        "API access": "Self-hosted OpenAI-compatible runtimes",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "ai21labs/AI21-Jamba2-3B (self-hosted; no AI21 SaaS endpoint)",
        "Computer use": "Via external agent/tool integrations",
        "Release date": "2026-01-08",
        "Context window": "256K",
        "Product access": "Hugging Face + private/VPC/on-prem deployment",
        "Knowledge cutoff": "Aug 22, 2024",
        "Weights / license": "Open source — Apache 2.0",
        "Reasoning / effort": "Instruction-following model; no separate hosted reasoning mode",
        "Image / vision input": "No",
        "Tool / function calling": "Yes in supported self-hosted runtimes"
      },
      "sources": [
        {
          "url": "https://docs.ai21.com/docs/jamba-foundation-models",
          "title": "AI21 Jamba model docs"
        },
        {
          "url": "https://huggingface.co/ai21labs/AI21-Jamba2-3B",
          "title": "Jamba2 3B official weights"
        }
      ],
      "notes": "Compact 3B hybrid Transformer–Mamba model with 28 layers, designed for on-device and enterprise agent stacks. AI21 publishes evaluation charts but not stable machine-readable numeric benchmark values, so no numeric rows are inferred.",
      "verified_at": "2026-09-07T09:28:38.262Z"
    },
    {
      "slug": "jamba2-mini",
      "name": "Jamba2 Mini",
      "developer": "AI21 Labs",
      "release_date": "2026-01-08T00:00:00.000Z",
      "access": "open_source",
      "comparison_data": {
        "Developer": "AI21 Labs",
        "API access": "AI21 SaaS + self-hosted",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "jamba-mini / jamba-mini-2-2026-01; ai21labs/AI21-Jamba2-Mini weights",
        "Computer use": "Via external agent/tool integrations",
        "Release date": "2026-01-08",
        "Context window": "256K",
        "Product access": "AI21 API + Hugging Face + private/VPC/on-prem deployment",
        "Knowledge cutoff": "Aug 22, 2024",
        "Weights / license": "Open source — Apache 2.0",
        "Reasoning / effort": "Enterprise reliability / steerability model; no separate reasoning mode",
        "Image / vision input": "No",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://docs.ai21.com/docs/jamba-foundation-models",
          "title": "AI21 Jamba model docs"
        },
        {
          "url": "https://huggingface.co/ai21labs/AI21-Jamba2-Mini",
          "title": "Jamba2 Mini official weights"
        }
      ],
      "notes": "52B-total / 12B-active hybrid Transformer–Mamba enterprise model focused on instruction following, grounding and long-context reliability. Numeric chart values are not inferred from images.",
      "verified_at": "2026-09-07T09:28:38.262Z"
    },
    {
      "slug": "kimi-k2-5",
      "name": "Kimi K2.5",
      "developer": "Moonshot AI",
      "release_date": "2026-01-27T00:00:00.000Z",
      "access": "open_weights",
      "comparison_data": {
        "Developer": "Moonshot AI",
        "API access": "Yes",
        "Max output": "64K recommended benchmark generation budget",
        "Text input": "Yes",
        "Audio input": "No native audio input documented",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "kimi-k2.5",
        "Audio output": "No",
        "Computer use": "Via Kimi Agent/Kimi Code tool environments",
        "Image output": "No",
        "Release date": "2026-01-27",
        "Video output": "No",
        "Context window": "256K",
        "Product access": "Kimi.com + Kimi App + Kimi API + Kimi Code",
        "Cached input / 1M": "¥0.70",
        "Input / 1M tokens": "¥4.00",
        "Weights / license": "Open weights — Modified MIT",
        "Output / 1M tokens": "¥21.00",
        "Reasoning / effort": "Thinking + non-thinking; Agent and Agent Swarm modes",
        "Humanity's Last Exam": "31.5% text/no tools / 51.8% text/with tools",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes",
        "MCP / tool-use benchmark": "HLE multimodal 21.3% no tools / 39.8% with tools"
      },
      "sources": [
        {
          "url": "https://www.kimi.com/en/blog/kimi-k2-5",
          "title": "Kimi K2.5 technical blog"
        },
        {
          "url": "https://platform.kimi.com/docs/models",
          "title": "Kimi API model list"
        },
        {
          "url": "https://platform.kimi.com/",
          "title": "Kimi API pricing"
        },
        {
          "url": "https://huggingface.co/moonshotai/Kimi-K2.5",
          "title": "Kimi K2.5 official weights"
        }
      ],
      "notes": "Native multimodal MoE with Agent Swarm support for up to 100 sub-agents and roughly 1,500 coordinated tool calls. Pricing is the current first-party Kimi API CNY rate.",
      "verified_at": "2026-09-07T09:28:38.262Z"
    },
    {
      "slug": "kimi-k2-6",
      "name": "Kimi K2.6",
      "developer": "Moonshot AI",
      "release_date": "2026-04-20T00:00:00.000Z",
      "access": "open_weights",
      "comparison_data": {
        "Developer": "Moonshot AI",
        "API access": "Yes",
        "Text input": "Yes",
        "Audio input": "No native audio input documented",
        "Text output": "Yes",
        "Video input": "No native video input documented",
        "API model ID": "kimi-k2.6",
        "Audio output": "No",
        "Computer use": "Via Kimi Agent / Agent Swarm",
        "Image output": "No",
        "Release date": "2026-04-20",
        "Video output": "No",
        "Context window": "256K tokens",
        "Product access": "Kimi + Kimi Agent + Kimi API + open weights",
        "Input / 1M tokens": "~$0.95 uncached",
        "Weights / license": "Open weights — Modified MIT",
        "Output / 1M tokens": "~$4",
        "Reasoning / effort": "Thinking and non-thinking modes",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://huggingface.co/moonshotai/Kimi-K2.6",
          "title": "Kimi K2.6 official model card"
        },
        {
          "url": "https://www.kimi.com/en/help/agent/agent-overview",
          "title": "Kimi Agent overview"
        },
        {
          "url": "https://www.kimi.com/en/help/agent/agent-swarm",
          "title": "Kimi Agent Swarm"
        }
      ],
      "notes": "1T-parameter MoE with 32B active parameters, native vision, 256K context and upgraded Agent Swarm. Hosted pricing is intentionally omitted because a directly attributable first-party K2.6 rate was not available in the verified source set.",
      "verified_at": "2026-09-07T04:00:13.944Z"
    },
    {
      "slug": "kimi-k3",
      "name": "Kimi K3",
      "developer": "Moonshot AI",
      "release_date": "2026-07-16T00:00:00.000Z",
      "access": "open_weights",
      "comparison_data": {
        "Developer": "Moonshot AI",
        "API access": "Yes",
        "Text input": "Yes",
        "Audio input": "No native audio input documented",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "kimi-k3",
        "Audio output": "No",
        "Computer use": "Via Kimi agent and coding harnesses",
        "Image output": "No",
        "Release date": "2026-07-16",
        "Video output": "No",
        "Context window": "1M tokens",
        "Product access": "Kimi + Kimi Work + Kimi Code + Kimi API + weights",
        "Cached input / 1M": "$0.30 cache hit",
        "Input / 1M tokens": "$3.00 cache miss",
        "Weights / license": "Open weights — Kimi K3 License",
        "Output / 1M tokens": "$15.00",
        "Reasoning / effort": "low / high / max",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://www.kimi.com/blog/kimi-k3",
          "title": "Kimi K3 research release"
        },
        {
          "url": "https://huggingface.co/moonshotai/Kimi-K3",
          "title": "Kimi K3 official model card"
        },
        {
          "url": "https://platform.moonshot.ai/docs/pricing",
          "title": "Kimi API pricing"
        }
      ],
      "notes": "2.8T-parameter MoE model with native vision and a 1M context window. Benchmark rows below use the direct Kimi evaluation where available rather than later cross-vendor reruns.",
      "verified_at": "2026-09-07T03:17:35.772Z"
    },
    {
      "slug": "minimax-m2-5",
      "name": "MiniMax M2.5",
      "developer": "MiniMax",
      "release_date": "2026-02-12T00:00:00.000Z",
      "access": "open_weights",
      "comparison_data": {
        "Developer": "MiniMax",
        "API access": "Yes",
        "BrowseComp": "76.3% with context management",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "MiniMax-M2.5",
        "Computer use": "Via agent/tool integrations",
        "Release date": "2026-02-12",
        "Context window": "205K combined input + output",
        "Product access": "MiniMax API + downloadable weights",
        "Cache write / 1M": "$0.375",
        "Cached input / 1M": "$0.03",
        "Input / 1M tokens": "$0.30",
        "Weights / license": "Open weights — Modified MIT",
        "Output / 1M tokens": "$1.20",
        "Reasoning / effort": "Interleaved thinking / agentic reasoning",
        "SWE-bench Verified": "80.2%",
        "Image / vision input": "No native image input",
        "Tool / function calling": "Yes",
        "MCP / tool-use benchmark": "Multi-SWE-Bench 51.3%"
      },
      "sources": [
        {
          "url": "https://www.minimax.io/news/minimax-m25",
          "title": "MiniMax M2.5 launch"
        },
        {
          "url": "https://platform.minimax.io/docs/guides/text-generation",
          "title": "MiniMax text generation docs"
        },
        {
          "url": "https://platform.minimax.io/docs/guides/pricing-paygo",
          "title": "MiniMax pay-as-you-go pricing"
        },
        {
          "url": "https://huggingface.co/MiniMaxAI/MiniMax-M2.5",
          "title": "MiniMax M2.5 weights"
        }
      ],
      "notes": "Legacy-active MiniMax M2 family model focused on real-world productivity, coding, tool use and search; standard API output speed is approximately 60 tok/s.",
      "verified_at": "2026-09-07T09:28:38.262Z"
    },
    {
      "slug": "minimax-m2-7",
      "name": "MiniMax M2.7",
      "developer": "MiniMax",
      "release_date": "2026-03-18T00:00:00.000Z",
      "access": "open_weights",
      "comparison_data": {
        "Developer": "MiniMax",
        "API access": "Yes",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "MiniMax-M2.7",
        "Computer use": "Via agent/tool integrations",
        "Release date": "2026-03-18",
        "SWE-bench Pro": "56.22%",
        "Context window": "205K combined input + output",
        "Product access": "MiniMax API + Token Plan + downloadable weights",
        "Cache write / 1M": "$0.375",
        "Cached input / 1M": "$0.06",
        "Input / 1M tokens": "$0.30",
        "Weights / license": "Open weights — non-commercial custom license; commercial authorization required",
        "Output / 1M tokens": "$1.20",
        "Reasoning / effort": "Interleaved thinking / recursive self-improvement training",
        "Terminal-Bench 2.1": "57.0% (Terminal Bench 2)",
        "Image / vision input": "No native image input",
        "Tool / function calling": "Yes",
        "MCP / tool-use benchmark": "VIBE-Pro 55.6%"
      },
      "sources": [
        {
          "url": "https://www.minimax.io/news/minimax-m27-en",
          "title": "MiniMax M2.7 launch"
        },
        {
          "url": "https://platform.minimax.io/docs/guides/text-generation",
          "title": "MiniMax text generation docs"
        },
        {
          "url": "https://platform.minimax.io/docs/guides/pricing-paygo",
          "title": "MiniMax pay-as-you-go pricing"
        },
        {
          "url": "https://huggingface.co/MiniMaxAI/MiniMax-M2.7",
          "title": "MiniMax M2.7 weights"
        }
      ],
      "notes": "Current M2-series model before M3, focused on software engineering and professional office delivery. Model weights use a non-commercial license; hosted commercial API access remains available.",
      "verified_at": "2026-09-07T09:28:38.262Z"
    },
    {
      "slug": "minimax-m3",
      "name": "MiniMax M3",
      "developer": "MiniMax",
      "release_date": "2026-06-01T00:00:00.000Z",
      "access": "open_weights",
      "comparison_data": {
        "Developer": "MiniMax",
        "API access": "Yes",
        "BrowseComp": "83.5% (MiniMax launch evaluation)",
        "Text input": "Yes",
        "Audio input": "No native audio input documented",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "MiniMax-M3",
        "Audio output": "No",
        "Computer use": "Yes",
        "Image output": "No",
        "Release date": "2026-06-01",
        "Video output": "No",
        "Context window": "1M tokens",
        "Product access": "MiniMax Code + MiniMax API + weights",
        "Cached input / 1M": "$0.06 <=512K / $0.12 >512K–1M",
        "Input / 1M tokens": "$0.30 <=512K / $0.60 >512K–1M",
        "Weights / license": "Open weights — MiniMax Community License",
        "Output / 1M tokens": "$1.20 <=512K / $2.40 >512K–1M",
        "Reasoning / effort": "Thinking can be enabled or disabled",
        "Image / vision input": "Yes",
        "Long-context surcharge": ">512K uses the long-context tier",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://www.minimax.io/news/minimax-m3",
          "title": "MiniMax M3"
        },
        {
          "url": "https://huggingface.co/MiniMaxAI/MiniMax-M3",
          "title": "MiniMax M3 official weights"
        },
        {
          "url": "https://www.minimax.io/platform/pricing",
          "title": "MiniMax Token Plan pricing"
        }
      ],
      "notes": "Native multimodal agent model with roughly 428B total and 23B active parameters. Current hosted prices reflect MiniMax's permanent 50% Token Plan rate.",
      "verified_at": "2026-09-07T03:17:35.772Z"
    },
    {
      "slug": "mistral-medium-3-5",
      "name": "Mistral Medium 3.5",
      "developer": "Mistral AI",
      "release_date": "2026-04-28T00:00:00.000Z",
      "access": "open_weights",
      "comparison_data": {
        "Developer": "Mistral AI",
        "API access": "Yes",
        "Text input": "Yes",
        "Audio input": "No native audio input documented",
        "Text output": "Yes",
        "Video input": "No native video input documented",
        "API model ID": "mistral-medium-3-5",
        "Audio output": "No",
        "Computer use": "Via long-horizon agent integrations",
        "Image output": "No",
        "Release date": "2026-04-28",
        "Video output": "No",
        "Context window": "256K tokens",
        "Product access": "Le Chat + Vibe + Mistral API + weights",
        "Cached input / 1M": "$0.15",
        "Input / 1M tokens": "$1.50",
        "Weights / license": "Open weights — Modified MIT",
        "Output / 1M tokens": "$7.50",
        "Reasoning / effort": "Configurable reasoning effort",
        "SWE-bench Verified": "77.6%",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes; function calling, agents, structured outputs and batching",
        "MCP / tool-use benchmark": "τ³-Telecom 91.4%"
      },
      "sources": [
        {
          "url": "https://docs.mistral.ai/models/mistral-medium-3-5-26-04",
          "title": "Mistral Medium 3.5 docs"
        },
        {
          "url": "https://mistral.ai/news/vibe-remote-agents-mistral-medium-3-5/",
          "title": "Remote agents in Vibe powered by Mistral Medium 3.5"
        },
        {
          "url": "https://docs.mistral.ai/inference/pricing",
          "title": "Mistral API pricing"
        }
      ],
      "notes": "Dense 128B multimodal model optimized for long-horizon coding and agent work. Current docs mark the v26.04 release GA under a Modified MIT license.",
      "verified_at": "2026-09-07T04:36:59.464Z"
    },
    {
      "slug": "mistral-small-4",
      "name": "Mistral Small 4",
      "developer": "Mistral AI",
      "release_date": "2026-03-16T00:00:00.000Z",
      "access": "open_source",
      "comparison_data": {
        "Developer": "Mistral AI",
        "API access": "Yes",
        "Text input": "Yes",
        "Audio input": "No native audio input documented",
        "Text output": "Yes",
        "Video input": "No native video input documented",
        "API model ID": "mistral-small-2603",
        "Audio output": "No",
        "Computer use": "Via agent integrations; no native computer-use API documented",
        "Image output": "No",
        "Release date": "2026-03-16",
        "Video output": "No",
        "Context window": "256K tokens",
        "Product access": "Mistral API + downloadable weights",
        "Cached input / 1M": "$0.015",
        "Input / 1M tokens": "$0.15",
        "Weights / license": "Open source — Apache 2.0",
        "Output / 1M tokens": "$0.60",
        "Reasoning / effort": "Configurable reasoning effort",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes; function calling, built-in tools, agents and structured outputs"
      },
      "sources": [
        {
          "url": "https://mistral.ai/news/mistral-small-4/",
          "title": "Introducing Mistral Small 4"
        },
        {
          "url": "https://docs.mistral.ai/models/mistral-small-4-0-26-03",
          "title": "Mistral Small 4 docs"
        },
        {
          "url": "https://docs.mistral.ai/inference/pricing",
          "title": "Mistral API pricing"
        }
      ],
      "notes": "119B-total / 6.5B-active hybrid model unifying instruct, reasoning, multimodal and agentic-coding capabilities.",
      "verified_at": "2026-09-07T04:36:59.464Z"
    },
    {
      "slug": "muse-glimmer",
      "name": "Muse Glimmer",
      "developer": "Meta",
      "release_date": "2026-08-10T00:00:00.000Z",
      "access": "open_source",
      "comparison_data": {
        "Developer": "Meta",
        "API access": "Self-hosted and third-party serving",
        "Text input": "Yes",
        "Audio input": "No native audio support documented",
        "Text output": "Yes",
        "Video input": "No native video support documented",
        "Computer use": "Via screenshot-driven/agent scaffolds; not a hosted native computer-use tool",
        "Release date": "2026-08-10",
        "Context window": "131K+ tokens",
        "Product access": "Downloadable weights + agent scaffolds such as OpenClaw",
        "Weights / license": "Open source — Apache 2.0",
        "Reasoning / effort": "Controllable reasoning effort",
        "Image / vision input": "Yes; interleaved text/image input",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://research.meta.ai/blog/introducing-muse-glimmer-open-agentic-model",
          "title": "Introducing Muse Glimmer"
        }
      ],
      "notes": "30B local-agent model, 100+ languages. Meta published weights under Apache 2.0; a 4-bit K-Quant build is under 20GB. The launch report discusses long-context memory but does not disclose one canonical token limit, so Context window remains unset.",
      "verified_at": "2026-09-06T16:49:38.513Z"
    },
    {
      "slug": "muse-spark",
      "name": "Muse Spark",
      "developer": "Meta",
      "release_date": "2026-04-08T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Meta",
        "API access": "Private preview at launch",
        "Text input": "Yes",
        "Text output": "Yes",
        "Release date": "2026-04-08",
        "Product access": "Meta AI + meta.ai",
        "Weights / license": "Proprietary",
        "Reasoning / effort": "Native reasoning; Contemplating mode uses parallel multi-agent test-time reasoning",
        "Humanity's Last Exam": "58 (Contemplating mode)",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes",
        "MCP / tool-use benchmark": "FrontierScience Research 38"
      },
      "sources": [
        {
          "url": "https://ai.meta.com/blog/introducing-muse-spark-msl/",
          "title": "Introducing Muse Spark"
        }
      ],
      "notes": "First Muse model. Meta described native multimodal reasoning, visual chain of thought, tool use and multi-agent orchestration. The launch post does not disclose a precise audio/video API modality matrix or token limits, so those fields remain unset.",
      "verified_at": "2026-09-06T16:49:38.513Z"
    },
    {
      "slug": "muse-spark-1-1",
      "name": "Muse Spark 1.1",
      "developer": "Meta",
      "release_date": "2026-07-09T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Meta",
        "API access": "Meta Model API public preview",
        "Text input": "Yes",
        "Audio input": "Yes",
        "Text output": "Yes",
        "Video input": "Yes",
        "Computer use": "Yes",
        "Release date": "2026-07-09",
        "Context window": "1M tokens",
        "Product access": "Meta AI + Muse Code + Meta Model API",
        "Weights / license": "Proprietary",
        "Reasoning / effort": "Thinking mode; long-horizon planning with compaction and multi-agent orchestration",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes; native tools, MCP and custom skills"
      },
      "sources": [
        {
          "url": "https://research.meta.ai/blog/introducing-muse-spark-meta-model-api",
          "title": "Muse Spark 1.1 and Meta Model API"
        }
      ],
      "notes": "Meta documents 1M context and perception across images, video and audio for the 1.1 agentic model. No first-party token price was disclosed in the launch material reviewed, so pricing is intentionally blank.",
      "verified_at": "2026-09-06T16:49:38.513Z"
    },
    {
      "slug": "muse-spark-1-2",
      "name": "Muse Spark 1.2",
      "developer": "Meta",
      "release_date": "2026-08-05T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Meta",
        "API access": "Meta Model API",
        "Text input": "Yes",
        "Audio input": "Yes",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "muse-spark-1.2",
        "Computer use": "Yes — through agentic computer/code harnesses",
        "Release date": "2026-08-05",
        "Primary focus": "Coding-focused update: code generation, complex debugging, codebase understanding and end-to-end developer workflows",
        "Context window": "1M tokens",
        "Product access": "Muse Code + Meta Model API + Meta AI",
        "Cached input / 1M": "$0.15",
        "Input / 1M tokens": "$1.25",
        "Long-horizon work": "Trained on whole-repository generation, large end-to-end projects and auto-research; uses planning, goal conditioning and context compaction to sustain progress",
        "Weights / license": "Proprietary; open weights were announced but not released for 1.2",
        "Output / 1M tokens": "$4.25",
        "Reasoning / effort": "xhigh used in Meta's 1.2 launch evaluations; long-horizon reasoning with compaction and subagents",
        "Agent orchestration": "Co-trained with Muse Code; coordinates persistent async subagents that retain task context across a session",
        "Image / vision input": "Yes",
        "File / document input": "Yes — multimodal files and PDFs supported in Meta Model API workflows",
        "Tool / function calling": "Yes — native tools, parallel agent workflows and Muse Code integration"
      },
      "sources": [
        {
          "url": "https://developer.meta.com/ai/models/muse-spark/",
          "title": "Muse Spark model card"
        },
        {
          "url": "https://research.meta.ai/blog/introducing-muse-code-and-muse-spark-1-2",
          "title": "Introducing Muse Code and Muse Spark 1.2"
        },
        {
          "url": "https://research.meta.ai/static/muse-spark-1-2-methodology",
          "title": "Muse Spark 1.2 evaluation methodology"
        },
        {
          "url": "https://research.meta.ai/blog/multimodal-intelligence-of-muse-spark-1-2",
          "title": "Multimodal intelligence of Muse Spark 1.2"
        }
      ],
      "notes": "Muse Spark 1.2 is a coding-focused Muse release co-trained with Muse Code. Meta documents persistent async subagents, replay-safe long-running workflows, whole-repository work, compaction, multimodal input and expanded Meta Model API access. Pricing/context fields reflect the Meta Model API model card; benchmark provenance remains normalized separately.",
      "verified_at": "2026-09-07T10:52:03.233Z"
    },
    {
      "slug": "muse-spark-1-3",
      "name": "Muse Spark 1.3",
      "developer": "Meta",
      "release_date": "2026-09-02T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Meta",
        "API access": "Meta Model API",
        "Text input": "Yes",
        "Audio input": "Yes — audio-file editing workflow demonstrated; exact low-level codec matrix not separately disclosed",
        "Text output": "Yes",
        "Video input": "Yes — multimodal file workflows",
        "API model ID": "muse-spark-1.3",
        "Computer use": "Yes — long-horizon agentic workflows",
        "Release date": "2026-09-02",
        "Primary focus": "Long-horizon agentic and coding work with improved real-world usability",
        "Context window": "1M tokens",
        "Product access": "Muse Code + Meta Model API + Meta AI",
        "Cached input / 1M": "$0.15",
        "Input / 1M tokens": "$1.25",
        "Long-horizon work": "Sustains longer work, generates context across messy/conflicting sources, corrects planning gaps, preserves detailed constraints and tracks multiple workflows in one long thread",
        "Weights / license": "Proprietary; Meta lists open weights as a future release",
        "Output / 1M tokens": "$4.25",
        "Reasoning / effort": "xhigh + max; Meta's launch scorecard uses max for 1.3 while the 1.2 comparison column uses xhigh",
        "Safety / approvals": "Improved prompt-injection/adversarial robustness and better calibration around irreversible actions",
        "User collaboration": "Asks clarifying questions for ambiguous prompts, requests help when stuck, adapts update frequency and confirms before consequential actions",
        "Agent orchestration": "Improved multitasking and routing of new prompts to the correct ongoing task; uses tools to build and maintain working context",
        "Image / vision input": "Yes — visual and heterogeneous file workflows",
        "File / document input": "Yes — documents, spreadsheets, CAD/STEP and other heterogeneous files demonstrated",
        "Tool / function calling": "Yes",
        "Efficiency / generation change": "In Meta engineer comparisons vs 1.2: ~20% fewer tool calls and ~25% fewer tokens for coding work"
      },
      "sources": [
        {
          "url": "https://developer.meta.com/ai/models/muse-spark/",
          "title": "Muse Spark model card"
        },
        {
          "url": "https://research.meta.ai/blog/introducing-muse-spark-1-3",
          "title": "Introducing Muse Spark 1.3"
        },
        {
          "url": "https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology",
          "title": "Muse Spark 1.3 evaluation methodology"
        }
      ],
      "notes": "Muse Spark 1.3 focuses on long-horizon agentic/coding work, multi-workflow tracking, user collaboration and efficiency. Meta reports about 20% fewer tool calls and 25% fewer tokens than 1.2 in internal engineering comparisons. Launch scorecard results use max reasoning for 1.3 and xhigh for 1.2, so the reasoning-effort difference is preserved in normalized benchmark qualifiers.",
      "verified_at": "2026-09-07T10:52:03.233Z"
    },
    {
      "slug": "qwen3-5-0-8b",
      "name": "Qwen3.5-0.8B",
      "developer": "Alibaba / Qwen",
      "release_date": "2026-03-02T00:00:00.000Z",
      "access": "open_source",
      "comparison_data": {
        "Developer": "Alibaba / Qwen",
        "API access": "Downloadable weights; self-hosted OpenAI-compatible servers",
        "Text input": "Yes",
        "Audio input": "No native audio input",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "Qwen/Qwen3.5-0.8B",
        "Audio output": "No",
        "Computer use": "Via external agent/tool harnesses; no native computer-use API",
        "Image output": "No",
        "Release date": "2026-03-02",
        "Video output": "No",
        "Context window": "262K native; extensible to ~1.01M",
        "Product access": "Qwen Chat + downloadable weights",
        "Weights / license": "Open source — Apache 2.0",
        "Reasoning / effort": "Thinking and non-thinking modes",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://huggingface.co/Qwen/Qwen3.5-0.8B",
          "title": "Qwen3.5-0.8B official weights"
        },
        {
          "url": "https://qwen.ai/blog?id=qwen3.5",
          "title": "Qwen3.5 launch"
        }
      ],
      "notes": "0.8B dense native multimodal model.",
      "verified_at": "2026-09-07T08:16:00.818Z"
    },
    {
      "slug": "qwen3-5-122b-a10b",
      "name": "Qwen3.5-122B-A10B",
      "developer": "Alibaba / Qwen",
      "release_date": "2026-02-24T00:00:00.000Z",
      "access": "open_source",
      "comparison_data": {
        "Developer": "Alibaba / Qwen",
        "API access": "Downloadable weights; self-hosted OpenAI-compatible servers",
        "Text input": "Yes",
        "Audio input": "No native audio input",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "Qwen/Qwen3.5-122B-A10B",
        "Audio output": "No",
        "Computer use": "Via external agent/tool harnesses; no native computer-use API",
        "Image output": "No",
        "Release date": "2026-02-24",
        "Video output": "No",
        "Context window": "262K native; extensible to ~1.01M",
        "Product access": "Qwen Chat + downloadable weights",
        "Weights / license": "Open source — Apache 2.0",
        "Reasoning / effort": "Thinking and non-thinking modes",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://huggingface.co/Qwen/Qwen3.5-122B-A10B",
          "title": "Qwen3.5-122B-A10B official weights"
        },
        {
          "url": "https://qwen.ai/blog?id=qwen3.5",
          "title": "Qwen3.5 launch"
        }
      ],
      "notes": "122B total / 10B active native multimodal model.",
      "verified_at": "2026-09-07T08:16:00.818Z"
    },
    {
      "slug": "qwen3-5-27b",
      "name": "Qwen3.5-27B",
      "developer": "Alibaba / Qwen",
      "release_date": "2026-02-24T00:00:00.000Z",
      "access": "open_source",
      "comparison_data": {
        "Developer": "Alibaba / Qwen",
        "API access": "Downloadable weights; self-hosted OpenAI-compatible servers",
        "Text input": "Yes",
        "Audio input": "No native audio input",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "Qwen/Qwen3.5-27B",
        "Audio output": "No",
        "Computer use": "Via external agent/tool harnesses; no native computer-use API",
        "Image output": "No",
        "Release date": "2026-02-24",
        "Video output": "No",
        "Context window": "262K native; extensible to ~1.01M",
        "Product access": "Qwen Chat + downloadable weights",
        "Weights / license": "Open source — Apache 2.0",
        "Reasoning / effort": "Thinking and non-thinking modes",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://huggingface.co/Qwen/Qwen3.5-27B",
          "title": "Qwen3.5-27B official weights"
        },
        {
          "url": "https://qwen.ai/blog?id=qwen3.5",
          "title": "Qwen3.5 launch"
        }
      ],
      "notes": "27B dense native multimodal model.",
      "verified_at": "2026-09-07T08:16:00.818Z"
    },
    {
      "slug": "qwen3-5-2b",
      "name": "Qwen3.5-2B",
      "developer": "Alibaba / Qwen",
      "release_date": "2026-03-02T00:00:00.000Z",
      "access": "open_source",
      "comparison_data": {
        "Developer": "Alibaba / Qwen",
        "API access": "Downloadable weights; self-hosted OpenAI-compatible servers",
        "Text input": "Yes",
        "Audio input": "No native audio input",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "Qwen/Qwen3.5-2B",
        "Audio output": "No",
        "Computer use": "Via external agent/tool harnesses; no native computer-use API",
        "Image output": "No",
        "Release date": "2026-03-02",
        "Video output": "No",
        "Context window": "262K native; extensible to ~1.01M",
        "Product access": "Qwen Chat + downloadable weights",
        "Weights / license": "Open source — Apache 2.0",
        "Reasoning / effort": "Thinking and non-thinking modes",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://huggingface.co/Qwen/Qwen3.5-2B",
          "title": "Qwen3.5-2B official weights"
        },
        {
          "url": "https://qwen.ai/blog?id=qwen3.5",
          "title": "Qwen3.5 launch"
        }
      ],
      "notes": "2B dense native multimodal model.",
      "verified_at": "2026-09-07T08:16:00.818Z"
    },
    {
      "slug": "qwen3-5-35b-a3b",
      "name": "Qwen3.5-35B-A3B",
      "developer": "Alibaba / Qwen",
      "release_date": "2026-02-24T00:00:00.000Z",
      "access": "open_source",
      "comparison_data": {
        "Developer": "Alibaba / Qwen",
        "API access": "Downloadable weights; self-hosted OpenAI-compatible servers",
        "Text input": "Yes",
        "Audio input": "No native audio input",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "Qwen/Qwen3.5-35B-A3B",
        "Audio output": "No",
        "Computer use": "Via external agent/tool harnesses; no native computer-use API",
        "Image output": "No",
        "Release date": "2026-02-24",
        "Video output": "No",
        "Context window": "262K native; extensible to ~1.01M",
        "Product access": "Qwen Chat + downloadable weights",
        "Weights / license": "Open source — Apache 2.0",
        "Reasoning / effort": "Thinking and non-thinking modes",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://huggingface.co/Qwen/Qwen3.5-35B-A3B",
          "title": "Qwen3.5-35B-A3B official weights"
        },
        {
          "url": "https://qwen.ai/blog?id=qwen3.5",
          "title": "Qwen3.5 launch"
        }
      ],
      "notes": "35B total / 3B active native multimodal model.",
      "verified_at": "2026-09-07T08:16:00.818Z"
    },
    {
      "slug": "qwen3-5-397b-a17b",
      "name": "Qwen3.5-397B-A17B",
      "developer": "Alibaba / Qwen",
      "release_date": "2026-02-15T00:00:00.000Z",
      "access": "open_source",
      "comparison_data": {
        "Developer": "Alibaba / Qwen",
        "API access": "Downloadable weights; self-hosted OpenAI-compatible servers",
        "Text input": "Yes",
        "Audio input": "No native audio input",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "Qwen/Qwen3.5-397B-A17B",
        "Audio output": "No",
        "Computer use": "Via external agent/tool harnesses; no native computer-use API",
        "Image output": "No",
        "Release date": "2026-02-15",
        "Video output": "No",
        "Context window": "262K native; extensible to ~1.01M",
        "Product access": "Qwen Chat + downloadable weights",
        "Weights / license": "Open source — Apache 2.0",
        "Reasoning / effort": "Thinking and non-thinking modes",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://qwen.ai/blog?id=qwen3.5",
          "title": "Qwen3.5-397B-A17B official weights"
        },
        {
          "url": "https://qwen.ai/blog?id=qwen3.5",
          "title": "Qwen3.5 launch"
        }
      ],
      "notes": "397B-total / 17B-active native multimodal MoE; hybrid Gated DeltaNet + gated attention.",
      "verified_at": "2026-09-07T08:16:00.818Z"
    },
    {
      "slug": "qwen3-5-4b",
      "name": "Qwen3.5-4B",
      "developer": "Alibaba / Qwen",
      "release_date": "2026-03-02T00:00:00.000Z",
      "access": "open_source",
      "comparison_data": {
        "Developer": "Alibaba / Qwen",
        "API access": "Downloadable weights; self-hosted OpenAI-compatible servers",
        "Text input": "Yes",
        "Audio input": "No native audio input",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "Qwen/Qwen3.5-4B",
        "Audio output": "No",
        "Computer use": "Via external agent/tool harnesses; no native computer-use API",
        "Image output": "No",
        "Release date": "2026-03-02",
        "Video output": "No",
        "Context window": "262K native; extensible to ~1.01M",
        "Product access": "Qwen Chat + downloadable weights",
        "Weights / license": "Open source — Apache 2.0",
        "Reasoning / effort": "Thinking and non-thinking modes",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://huggingface.co/Qwen/Qwen3.5-4B",
          "title": "Qwen3.5-4B official weights"
        },
        {
          "url": "https://qwen.ai/blog?id=qwen3.5",
          "title": "Qwen3.5 launch"
        }
      ],
      "notes": "4B dense native multimodal model.",
      "verified_at": "2026-09-07T08:16:00.818Z"
    },
    {
      "slug": "qwen3-5-9b",
      "name": "Qwen3.5-9B",
      "developer": "Alibaba / Qwen",
      "release_date": "2026-03-02T00:00:00.000Z",
      "access": "open_source",
      "comparison_data": {
        "Developer": "Alibaba / Qwen",
        "API access": "Downloadable weights; self-hosted OpenAI-compatible servers",
        "Text input": "Yes",
        "Audio input": "No native audio input",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "Qwen/Qwen3.5-9B",
        "Audio output": "No",
        "Computer use": "Via external agent/tool harnesses; no native computer-use API",
        "Image output": "No",
        "Release date": "2026-03-02",
        "Video output": "No",
        "Context window": "262K native; extensible to ~1.01M",
        "Product access": "Qwen Chat + downloadable weights",
        "Weights / license": "Open source — Apache 2.0",
        "Reasoning / effort": "Thinking and non-thinking modes",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://huggingface.co/Qwen/Qwen3.5-9B",
          "title": "Qwen3.5-9B official weights"
        },
        {
          "url": "https://qwen.ai/blog?id=qwen3.5",
          "title": "Qwen3.5 launch"
        }
      ],
      "notes": "9B dense native multimodal model.",
      "verified_at": "2026-09-07T08:16:00.818Z"
    },
    {
      "slug": "qwen3-5-flash",
      "name": "Qwen3.5-Flash",
      "developer": "Alibaba / Qwen",
      "release_date": "2026-02-23T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Alibaba / Qwen",
        "API access": "Alibaba Cloud Model Studio",
        "Max output": "64K",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "qwen3.5-flash",
        "Audio output": "No",
        "Computer use": "Via built-in/external tools; no native computer-use API",
        "Image output": "No",
        "Release date": "2026-02-23",
        "Video output": "No",
        "Context window": "1M",
        "Product access": "Qwen Studio + Alibaba Cloud Model Studio",
        "Input / 1M tokens": "$0.10",
        "Weights / license": "Proprietary hosted model",
        "Output / 1M tokens": "$0.40",
        "Reasoning / effort": "Thinking and non-thinking modes",
        "Image / vision input": "Yes",
        "Batch / flex discount": "50% batch inference discount",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://www.alibabacloud.com/help/en/model-studio/qwen3-5-flash",
          "title": "Qwen3.5-Flash official docs"
        },
        {
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "title": "Alibaba Model Studio pricing"
        }
      ],
      "notes": "Hosted low-cost Qwen3.5 model with function calling, structured output, web search and context caching.",
      "verified_at": "2026-09-07T08:16:01.059Z"
    },
    {
      "slug": "qwen3-5-plus",
      "name": "Qwen3.5-Plus",
      "developer": "Alibaba / Qwen",
      "release_date": "2026-02-15T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Alibaba / Qwen",
        "API access": "Alibaba Cloud Model Studio",
        "Max output": "64K",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "qwen3.5-plus",
        "Audio output": "No",
        "Computer use": "Via built-in/external tools; no native computer-use API",
        "Image output": "No",
        "Release date": "2026-02-15",
        "Video output": "No",
        "Context window": "1M",
        "Product access": "Qwen Studio + Alibaba Cloud Model Studio",
        "Input / 1M tokens": "$0.40 ≤256K / $0.50 >256K",
        "Weights / license": "Proprietary hosted model",
        "Output / 1M tokens": "$2.40 ≤256K / $3.00 >256K",
        "Reasoning / effort": "Thinking and non-thinking modes",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://qwen.ai/blog?id=qwen3.5",
          "title": "Qwen3.5-Plus official docs"
        },
        {
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "title": "Alibaba Model Studio pricing"
        }
      ],
      "notes": "Hosted Qwen3.5 flagship corresponding to the 397B-A17B family; includes official built-in tools and adaptive tool use.",
      "verified_at": "2026-09-07T08:16:01.059Z"
    },
    {
      "slug": "qwen3-6-27b",
      "name": "Qwen3.6-27B",
      "developer": "Alibaba / Qwen",
      "release_date": "2026-04-21T00:00:00.000Z",
      "access": "open_source",
      "comparison_data": {
        "MMLU-Pro": "86.2%",
        "Developer": "Alibaba / Qwen",
        "API access": "Yes",
        "Max output": "64K hosted",
        "Text input": "Yes",
        "Audio input": "No native audio input documented",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "qwen3.6-27b",
        "Audio output": "No",
        "Computer use": "Via agent/tool harnesses; no native computer-use API documented",
        "GPQA Diamond": "87.8%",
        "Image output": "No",
        "Release date": "2026-04-21",
        "Video output": "No",
        "LiveCodeBench": "83.9% (v6)",
        "SWE-bench Pro": "53.5%",
        "Context window": "262K hosted; open weights extensible to ~1.01M",
        "Product access": "Qwen Studio + Alibaba Model Studio + weights",
        "Input / 1M tokens": "$0.60 (Alibaba Model Studio Singapore)",
        "Weights / license": "Open source — Apache 2.0",
        "Output / 1M tokens": "$3.60 (Alibaba Model Studio Singapore)",
        "Reasoning / effort": "Thinking and non-thinking",
        "SWE-bench Verified": "77.2%",
        "Terminal-Bench 2.1": "59.3% (official model card evaluation)",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://qwen.ai/blog?id=qwen3.6-27b",
          "title": "Qwen3.6-27B launch"
        },
        {
          "url": "https://huggingface.co/Qwen/Qwen3.6-27B",
          "title": "Qwen3.6-27B official weights"
        },
        {
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "title": "Alibaba Model Studio pricing"
        }
      ],
      "notes": "Dense 27B multimodal model released for practical local deployment, with first-party agentic-coding evaluations and hosted API access.",
      "verified_at": "2026-09-07T04:36:58.081Z"
    },
    {
      "slug": "qwen3-6-35b-a3b",
      "name": "Qwen3.6-35B-A3B",
      "developer": "Alibaba / Qwen",
      "release_date": "2026-04-15T00:00:00.000Z",
      "access": "open_source",
      "comparison_data": {
        "Developer": "Alibaba / Qwen",
        "API access": "Yes",
        "Max output": "64K hosted",
        "Text input": "Yes",
        "Audio input": "No native audio input documented",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "qwen3.6-35b-a3b",
        "Audio output": "No",
        "Computer use": "Via agent/tool harnesses; no native computer-use API documented",
        "Image output": "No",
        "Release date": "2026-04-15",
        "Video output": "No",
        "Context window": "262K hosted; 262K native weights, extensible to ~1.01M",
        "Product access": "Qwen Studio + Alibaba Model Studio + weights",
        "Input / 1M tokens": "$0.375 (Alibaba Model Studio Singapore)",
        "Weights / license": "Open source — Apache 2.0",
        "Output / 1M tokens": "$2.25 (Alibaba Model Studio Singapore)",
        "Reasoning / effort": "Thinking and non-thinking; reasoning preservation supported",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://huggingface.co/Qwen/Qwen3.6-35B-A3B",
          "title": "Qwen3.6-35B-A3B official weights"
        },
        {
          "url": "https://www.alibabacloud.com/help/en/model-studio/qwen3-6-35b-a3b",
          "title": "Alibaba Model Studio qwen3.6-35b-a3b"
        },
        {
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "title": "Alibaba Model Studio pricing"
        }
      ],
      "notes": "35B-total / 3B-active multimodal MoE. The open checkpoint is natively 262,144 tokens and supports long-context extension; hosted limits/pricing use the Singapore international API.",
      "verified_at": "2026-09-07T04:36:58.081Z"
    },
    {
      "slug": "qwen3-6-flash",
      "name": "Qwen3.6-Flash",
      "developer": "Alibaba / Qwen",
      "release_date": "2026-04-16T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Alibaba / Qwen",
        "API access": "Alibaba Cloud Model Studio",
        "Max output": "64K",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "qwen3.6-flash",
        "Audio output": "No",
        "Computer use": "Via built-in/external tools; no native computer-use API",
        "Image output": "No",
        "Release date": "2026-04-16",
        "Video output": "No",
        "Context window": "1M",
        "Product access": "Qwen Studio + Alibaba Cloud Model Studio",
        "Input / 1M tokens": "$0.25 ≤256K / $1.00 >256K",
        "Weights / license": "Proprietary hosted model",
        "Output / 1M tokens": "$1.50 ≤256K / $4.00 >256K",
        "Reasoning / effort": "Thinking and non-thinking modes",
        "Image / vision input": "Yes",
        "Batch / flex discount": "50% batch inference discount",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://www.alibabacloud.com/help/en/model-studio/qwen3-6-flash",
          "title": "Qwen3.6-Flash official docs"
        },
        {
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "title": "Alibaba Model Studio pricing"
        }
      ],
      "notes": "Hosted multimodal Qwen3.6 Flash model.",
      "verified_at": "2026-09-07T08:16:01.059Z"
    },
    {
      "slug": "qwen3-6-max-preview",
      "name": "Qwen3.6-Max-Preview",
      "developer": "Alibaba / Qwen",
      "release_date": "2026-04-18T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Alibaba / Qwen",
        "API access": "Alibaba Cloud Model Studio",
        "Max output": "64K",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "qwen3.6-max-preview",
        "Audio output": "No",
        "Computer use": "Via built-in/external tools; no native computer-use API",
        "Image output": "No",
        "Release date": "2026-04-18",
        "Video output": "No",
        "Context window": "256K",
        "Product access": "Qwen Studio + Alibaba Cloud Model Studio",
        "Cache write / 1M": "$1.625 ≤128K / $2.50 >128K",
        "Cached input / 1M": "$0.13 explicit cache read ≤128K / $0.20 >128K",
        "Input / 1M tokens": "$1.30 ≤128K / $2.00 >128K",
        "Weights / license": "Proprietary hosted model",
        "Output / 1M tokens": "$7.80 ≤128K / $12.00 >128K",
        "Reasoning / effort": "Thinking and non-thinking modes",
        "Image / vision input": "No",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://qwen.ai/blog?id=qwen3.6-max-preview",
          "title": "Qwen3.6-Max-Preview official docs"
        },
        {
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "title": "Alibaba Model Studio pricing"
        }
      ],
      "notes": "Text-only proprietary preview focused on agentic coding and long-tail knowledge.",
      "verified_at": "2026-09-07T08:16:01.059Z"
    },
    {
      "slug": "qwen3-6-plus",
      "name": "Qwen3.6-Plus",
      "developer": "Alibaba / Qwen",
      "release_date": "2026-04-02T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Alibaba / Qwen",
        "API access": "Alibaba Cloud Model Studio",
        "Max output": "64K",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "qwen3.6-plus",
        "Audio output": "No",
        "Computer use": "Via built-in/external tools; no native computer-use API",
        "Image output": "No",
        "Release date": "2026-04-02",
        "Video output": "No",
        "Context window": "1M",
        "Product access": "Qwen Studio + Alibaba Cloud Model Studio",
        "Input / 1M tokens": "$0.50 ≤256K / $2.00 >256K",
        "Weights / license": "Proprietary hosted model",
        "Output / 1M tokens": "$3.00 ≤256K / $6.00 >256K",
        "Reasoning / effort": "Thinking and non-thinking modes",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://www.alibabacloud.com/help/en/model-studio/qwen3-6-plus",
          "title": "Qwen3.6-Plus official docs"
        },
        {
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "title": "Alibaba Model Studio pricing"
        }
      ],
      "notes": "Hosted multimodal Qwen3.6 flagship with built-in tools, structured output and web search.",
      "verified_at": "2026-09-07T08:16:01.059Z"
    },
    {
      "slug": "qwen3-7-max",
      "name": "Qwen3.7-Max",
      "developer": "Alibaba / Qwen",
      "release_date": "2026-05-20T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Alibaba / Qwen",
        "API access": "Alibaba Cloud Model Studio",
        "Max output": "128K",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "qwen3.7-max",
        "Audio output": "No",
        "Computer use": "Via built-in/external tools; no native computer-use API",
        "Image output": "No",
        "Release date": "2026-05-20",
        "Video output": "No",
        "Context window": "1M",
        "Product access": "Qwen Studio + Alibaba Cloud Model Studio",
        "Input / 1M tokens": "$2.50",
        "Weights / license": "Proprietary hosted model",
        "Output / 1M tokens": "$7.50",
        "Reasoning / effort": "Thinking and non-thinking modes",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://www.alibabacloud.com/help/en/model-studio/qwen3-7-max",
          "title": "Qwen3.7-Max official docs"
        },
        {
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "title": "Alibaba Model Studio pricing"
        }
      ],
      "notes": "Current Qwen3.7-Max points to the multimodal family; later snapshot qwen3.7-max-2026-06-08 adds image/video API support.",
      "verified_at": "2026-09-07T08:16:01.059Z"
    },
    {
      "slug": "qwen3-7-plus",
      "name": "Qwen3.7-Plus",
      "developer": "Alibaba / Qwen",
      "release_date": "2026-05-26T00:00:00.000Z",
      "access": "proprietary",
      "comparison_data": {
        "Developer": "Alibaba / Qwen",
        "API access": "Alibaba Cloud Model Studio",
        "Max output": "128K",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "qwen3.7-plus",
        "Audio output": "No",
        "Computer use": "Via built-in/external tools; no native computer-use API",
        "Image output": "No",
        "Release date": "2026-05-26",
        "Video output": "No",
        "Context window": "1M",
        "Product access": "Qwen Studio + Alibaba Cloud Model Studio",
        "Input / 1M tokens": "$0.40 ≤256K / $1.20 >256K",
        "Weights / license": "Proprietary hosted model",
        "Output / 1M tokens": "$1.60 ≤256K / $4.80 >256K",
        "Reasoning / effort": "Thinking and non-thinking modes",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://www.alibabacloud.com/help/en/model-studio/qwen3-7-plus",
          "title": "Qwen3.7-Plus official docs"
        },
        {
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "title": "Alibaba Model Studio pricing"
        }
      ],
      "notes": "Hosted multimodal Qwen3.7 Plus model; later Qwen3.8 launch provides cross-generation evaluation results.",
      "verified_at": "2026-09-07T08:16:01.059Z"
    },
    {
      "slug": "qwen3-8-2-4t-a95b",
      "name": "Qwen3.8-2.4T-A95B",
      "developer": "Alibaba / Qwen",
      "release_date": "2026-08-12T00:00:00.000Z",
      "access": "open_weights",
      "comparison_data": {
        "Developer": "Alibaba / Qwen",
        "API access": "Yes",
        "Max output": "128K hosted",
        "Text input": "Yes",
        "Audio input": "No",
        "Text output": "Yes",
        "Video input": "No",
        "API model ID": "qwen3.8-2.4t-a95b",
        "Audio output": "No",
        "Computer use": "Via tool/agent integrations; no native computer-use API documented",
        "Image output": "No",
        "Release date": "2026-08-12",
        "Video output": "No",
        "SWE-bench Pro": "67.7% (Qwen model card)",
        "Context window": "1M hosted; 262K native checkpoint, extensible to ~1.01M",
        "Product access": "Alibaba Model Studio + open weights",
        "Cache write / 1M": "$2.50 explicit cache creation",
        "Cached input / 1M": "$0.25 implicit / $0.17 explicit cache hit",
        "Input / 1M tokens": "$2.00 (Alibaba Model Studio Singapore)",
        "Weights / license": "Open weights — Qwen3.8-Max License",
        "Output / 1M tokens": "$6.00 (Alibaba Model Studio Singapore)",
        "Reasoning / effort": "Thinking and non-thinking",
        "Terminal-Bench 2.1": "86.6% (Qwen model card)",
        "Image / vision input": "No",
        "Tool / function calling": "Yes; function calling, structured output and web search"
      },
      "sources": [
        {
          "url": "https://www.alibabacloud.com/help/en/model-studio/qwen3-8-2-4t-a95b",
          "title": "Alibaba Model Studio qwen3.8-2.4t-a95b"
        },
        {
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "title": "Alibaba Model Studio pricing"
        },
        {
          "url": "https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B",
          "title": "Qwen3.8-2.4T-A95B official weights"
        }
      ],
      "notes": "2.4T-total / ~95B-active sparse MoE. Hosted Model Studio supports a full 1M context and context caching; the downloadable checkpoint uses a custom Qwen3.8-Max license rather than Apache 2.0.",
      "verified_at": "2026-09-07T04:36:58.081Z"
    },
    {
      "slug": "qwen3-8-27b",
      "name": "Qwen3.8-27B",
      "developer": "Alibaba / Qwen",
      "release_date": "2026-08-17T00:00:00.000Z",
      "access": "open_weights",
      "comparison_data": {
        "Developer": "Alibaba / Qwen",
        "API access": "Yes",
        "Max output": "128K tokens",
        "Text input": "Yes",
        "Audio input": "No native audio input documented",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "qwen3.8-27b",
        "Audio output": "No",
        "Computer use": "Via multimodal agent stacks",
        "Image output": "No",
        "Release date": "2026-08-17",
        "Video output": "No",
        "Context window": "1M tokens",
        "Product access": "Alibaba Model Studio + weights",
        "Cache write / 1M": "$0.625 explicit cache creation",
        "Cached input / 1M": "$0.10 implicit / $0.05 explicit cache read",
        "Input / 1M tokens": "$0.50 Alibaba Model Studio Singapore",
        "Weights / license": "Open weights — Apache 2.0",
        "Output / 1M tokens": "$3.00 Alibaba Model Studio Singapore",
        "Reasoning / effort": "Hybrid thinking / non-thinking; thinking enabled by default",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "title": "Qwen3.8-27B model docs"
        },
        {
          "url": "https://huggingface.co/Qwen/Qwen3.8-27B",
          "title": "Qwen3.8-27B official weights"
        }
      ],
      "notes": "27.3B dense native multimodal model. Hosted API supports a 1M context and 128K maximum output; pricing fields use Alibaba Model Studio Singapore international rates.",
      "verified_at": "2026-09-07T03:17:35.772Z"
    },
    {
      "slug": "qwen3-8-flash-next",
      "name": "Qwen3.8-Flash-Next",
      "developer": "Alibaba / Qwen",
      "release_date": "2026-08-26T00:00:00.000Z",
      "access": "open_weights",
      "comparison_data": {
        "Developer": "Alibaba / Qwen",
        "API access": "Yes via hosted qwen3.8-flash derivative",
        "Max output": "128K on hosted qwen3.8-flash",
        "Text input": "Yes",
        "Audio input": "No native audio input documented",
        "OSWorld 2.0": "52.3% partial / 19.4% binary",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "qwen3.8-flash (hosted derivative)",
        "Audio output": "No",
        "Computer use": "Via multimodal agent stacks",
        "DeepSWE v1.1": "58.7%",
        "Image output": "No",
        "Release date": "2026-08-26",
        "Video output": "No",
        "SWE-bench Pro": "62.5%",
        "Context window": "262K native; extensible to 1M",
        "Product access": "Open weights + hosted qwen3.8-flash",
        "Input / 1M tokens": "$0.16 hosted qwen3.8-flash launch rate",
        "Weights / license": "Open weights — Qwen Community License 1.0",
        "Output / 1M tokens": "$0.47 hosted qwen3.8-flash launch rate",
        "Reasoning / effort": "Thinking and non-thinking; thinking preservation supported",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes; hosted derivative includes built-in tools",
        "MCP / tool-use benchmark": "Toolathlon Verified 73.5%; AndroidWorld 84.5%"
      },
      "sources": [
        {
          "url": "https://qwen.ai/blog?id=qwen3.8-flash-next",
          "title": "Qwen3.8-Flash-Next launch"
        },
        {
          "url": "https://www.alibabacloud.com/help/en/model-studio/newly-released-models",
          "title": "Alibaba Model Studio model releases"
        }
      ],
      "notes": "Experimental open-weight Flash architecture: 125B main parameters plus 51B n-gram embedding parameters, with 6B active per token. Hosted pricing refers explicitly to the production qwen3.8-flash derivative.",
      "verified_at": "2026-09-07T04:36:58.081Z"
    },
    {
      "slug": "qwen3-8-max",
      "name": "Qwen3.8-Max",
      "developer": "Alibaba / Qwen",
      "release_date": "2026-08-02T00:00:00.000Z",
      "access": "open_weights",
      "comparison_data": {
        "Developer": "Alibaba / Qwen",
        "API access": "Yes",
        "Max output": "128K tokens",
        "Text input": "Yes",
        "Audio input": "No native audio input documented",
        "Text output": "Yes",
        "Video input": "Yes",
        "API model ID": "qwen3.8-max",
        "Audio output": "No",
        "Computer use": "Via multimodal agent stacks",
        "Image output": "No",
        "Release date": "2026-08-02",
        "Video output": "No",
        "Context window": "1M tokens",
        "Product access": "Alibaba Model Studio + Qwen weights",
        "Cache write / 1M": "$2.50 explicit cache creation",
        "Cached input / 1M": "$0.25 implicit / $0.17 explicit cache read",
        "Input / 1M tokens": "$2.00 Alibaba Model Studio Singapore",
        "Weights / license": "Open weights — Apache 2.0 counterpart Qwen3.8-2.4T-A95B",
        "Output / 1M tokens": "$6.00 Alibaba Model Studio Singapore",
        "Reasoning / effort": "Hybrid thinking / non-thinking; thinking enabled by default",
        "Image / vision input": "Yes",
        "Tool / function calling": "Yes"
      },
      "sources": [
        {
          "url": "https://www.alibabacloud.com/help/en/model-studio/newly-released-models",
          "title": "Qwen3.8-Max release"
        },
        {
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "title": "Qwen3.8-Max model docs"
        },
        {
          "url": "https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B",
          "title": "Qwen3.8 open model card"
        }
      ],
      "notes": "Hosted flagship model with a 2.4T-parameter open-weight counterpart. Current hosted snapshot is qwen3.8-max-0902; pricing fields use Alibaba Model Studio Singapore international rates.",
      "verified_at": "2026-09-07T03:17:35.772Z"
    }
  ],
  "benchmarks": [
    {
      "id": "tq-20260907-fable5-cursor32",
      "model_slug": "claude-fable-5",
      "category": "coding",
      "benchmark_name": "CursorBench",
      "benchmark_version": "3.2",
      "score_numeric": 70.5,
      "score_display": "70.5%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "max",
      "harness": "CursorBench 3.2",
      "evaluator": "SpaceXAI",
      "evaluation_date": "2026-08-12T00:00:00.000Z",
      "source": "https://x.ai/news/grok-4-6",
      "notes": "Cross-vendor comparison reported by SpaceXAI."
    },
    {
      "id": "tq-20260907-fable5-deepswe11",
      "model_slug": "claude-fable-5",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": null,
      "score_numeric": 70,
      "score_display": "70.0%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "max",
      "harness": "mini-swe-agent",
      "evaluator": "SpaceXAI",
      "evaluation_date": "2026-08-12T00:00:00.000Z",
      "source": "https://x.ai/news/grok-4-6",
      "notes": "Cross-vendor comparison reported by SpaceXAI."
    },
    {
      "id": "tq-20260907-fable5-frontiercode-ext",
      "model_slug": "claude-fable-5",
      "category": "coding",
      "benchmark_name": "FrontierCode 1.1 Extended",
      "benchmark_version": null,
      "score_numeric": 63.6,
      "score_display": "63.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "max",
      "harness": null,
      "evaluator": "SpaceXAI",
      "evaluation_date": "2026-08-12T00:00:00.000Z",
      "source": "https://x.ai/news/grok-4-6",
      "notes": "Cross-vendor comparison reported by SpaceXAI."
    },
    {
      "id": "tq-20260907-fable5-gdpv2",
      "model_slug": "claude-fable-5",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA v2",
      "benchmark_version": null,
      "score_numeric": 1741,
      "score_display": "1741 Elo",
      "score_unit": "Elo",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "Artificial Analysis",
      "evaluator": "SpaceXAI",
      "evaluation_date": "2026-08-12T00:00:00.000Z",
      "source": "https://x.ai/news/grok-4-6",
      "notes": "Cross-vendor comparison; underlying evaluation is Artificial Analysis."
    },
    {
      "id": "tq-20260907-fable5-swepro",
      "model_slug": "claude-fable-5",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": "Public",
      "score_numeric": 80.4,
      "score_display": "80.4%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "max",
      "harness": null,
      "evaluator": "SpaceXAI",
      "evaluation_date": "2026-07-16T00:00:00.000Z",
      "source": "https://x.ai/news/grok-4-5",
      "notes": "Cross-vendor comparison reported by SpaceXAI."
    },
    {
      "id": "tq-20260907-fable5-terminal21",
      "model_slug": "claude-fable-5",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": null,
      "score_numeric": 84.3,
      "score_display": "84.3%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "max",
      "harness": "Terminus-2",
      "evaluator": "SpaceXAI",
      "evaluation_date": "2026-07-16T00:00:00.000Z",
      "source": "https://x.ai/news/grok-4-5",
      "notes": "Cross-vendor comparison reported by SpaceXAI."
    },
    {
      "id": "tq-20260907-fable5-terminal30",
      "model_slug": "claude-fable-5",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 3.0",
      "benchmark_version": null,
      "score_numeric": 34.1,
      "score_display": "34.1%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "max",
      "harness": null,
      "evaluator": "SpaceXAI",
      "evaluation_date": "2026-08-12T00:00:00.000Z",
      "source": "https://x.ai/news/grok-4-6",
      "notes": "Cross-vendor comparison reported by SpaceXAI."
    },
    {
      "id": "tq-20260907-fable5-terminal-science",
      "model_slug": "claude-fable-5",
      "category": "coding",
      "benchmark_name": "Terminal-Bench Science 0.1",
      "benchmark_version": null,
      "score_numeric": 24.7,
      "score_display": "24.7%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Anthropic reproduction; Claude Code",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-09-01T00:00:00.000Z",
      "source": "https://www.anthropic.com/claude/fable",
      "notes": "Anthropic reproduction of the public leaderboard setup; public leaderboard score cited as 21.4%."
    },
    {
      "id": "claude-fable-5-1-automationbench-31-4-none-none-none-anthropic",
      "model_slug": "claude-fable-5-1",
      "category": "agentic_computer_use",
      "benchmark_name": "AutomationBench",
      "benchmark_version": null,
      "score_numeric": 31.4,
      "score_display": "31.4",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Anthropic",
      "evaluation_date": null,
      "source": "https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card",
      "notes": null
    },
    {
      "id": "claude-fable-5-1-cursorbench-3-2-0-73-4-none-none-none-anthropic",
      "model_slug": "claude-fable-5-1",
      "category": "coding",
      "benchmark_name": "CursorBench",
      "benchmark_version": "3.2.0",
      "score_numeric": 73.4,
      "score_display": "73.4",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Anthropic",
      "evaluation_date": null,
      "source": "https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card",
      "notes": null
    },
    {
      "id": "openai-20260903-fable51-deepswe-v1-1",
      "model_slug": "claude-fable-5-1",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": null,
      "score_numeric": 67.4,
      "score_display": "67.4%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "GPT-6 Astra launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-09-03T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-6-astra/",
      "notes": "Cross-vendor comparison reported by OpenAI; preserve evaluator/harness when comparing."
    },
    {
      "id": "openai-20260903-fable51-frontiercode-1-1-extended",
      "model_slug": "claude-fable-5-1",
      "category": "coding",
      "benchmark_name": "FrontierCode 1.1 Extended",
      "benchmark_version": null,
      "score_numeric": 63.6,
      "score_display": "63.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "GPT-6 Astra launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-09-03T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-6-astra/",
      "notes": "Cross-vendor comparison reported by OpenAI; preserve evaluator/harness when comparing."
    },
    {
      "id": "openai-20260903-fable51-frontiercode-1-1-main",
      "model_slug": "claude-fable-5-1",
      "category": "coding",
      "benchmark_name": "FrontierCode 1.1 Main",
      "benchmark_version": null,
      "score_numeric": 50.9,
      "score_display": "50.9%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "GPT-6 Astra launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-09-03T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-6-astra/",
      "notes": "Cross-vendor comparison reported by OpenAI; preserve evaluator/harness when comparing."
    },
    {
      "id": "openai-20260903-fable51-frontiermath-tier-4-v2",
      "model_slug": "claude-fable-5-1",
      "category": "math_reasoning",
      "benchmark_name": "FrontierMath Tier 4 (v2)",
      "benchmark_version": null,
      "score_numeric": 87.8,
      "score_display": "87.8%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "GPT-6 Astra launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-09-03T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-6-astra/",
      "notes": "Cross-vendor comparison reported by OpenAI; preserve evaluator/harness when comparing."
    },
    {
      "id": "openai-20260903-fable51-gpqa-diamond",
      "model_slug": "claude-fable-5-1",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 93.7,
      "score_display": "93.7%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "GPT-6 Astra launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-09-03T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-6-astra/",
      "notes": "Cross-vendor comparison reported by OpenAI; preserve evaluator/harness when comparing."
    },
    {
      "id": "claude-fable-5-1-humanity-s-last-exam-60-9-false-none-none-anthropic",
      "model_slug": "claude-fable-5-1",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 60.9,
      "score_display": "60.9",
      "score_unit": "%",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Anthropic",
      "evaluation_date": null,
      "source": "https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card",
      "notes": null
    },
    {
      "id": "claude-fable-5-1-humanity-s-last-exam-65-0-true-none-none-anthropic-tools",
      "model_slug": "claude-fable-5-1",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 65,
      "score_display": "65.0",
      "score_unit": "%",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Anthropic",
      "evaluation_date": null,
      "source": "https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card",
      "notes": null
    },
    {
      "id": "openai-20260903-fable51-humanity-s-last-exam",
      "model_slug": "claude-fable-5-1",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 65,
      "score_display": "65.0%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": "GPT-6 Astra launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-09-03T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-6-astra/",
      "notes": "Cross-vendor comparison reported by OpenAI; preserve evaluator/harness when comparing."
    },
    {
      "id": "claude-fable-5-1-osworld-2-0-partial-77-9-none-none-none-anthropic",
      "model_slug": "claude-fable-5-1",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld",
      "benchmark_version": "2.0 partial",
      "score_numeric": 77.9,
      "score_display": "77.9",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Anthropic",
      "evaluation_date": null,
      "source": "https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card",
      "notes": null
    },
    {
      "id": "claude-fable-5-1-osworld-2-0-strict-41-7-none-none-none-anthropic",
      "model_slug": "claude-fable-5-1",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld",
      "benchmark_version": "2.0 strict",
      "score_numeric": 41.7,
      "score_display": "41.7",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Anthropic",
      "evaluation_date": null,
      "source": "https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card",
      "notes": null
    },
    {
      "id": "claude-fable-5-1-terminal-bench-4-0-55-8-none-none-none-anthropic",
      "model_slug": "claude-fable-5-1",
      "category": "coding",
      "benchmark_name": "Terminal-Bench",
      "benchmark_version": "4.0",
      "score_numeric": 55.8,
      "score_display": "55.8",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Anthropic",
      "evaluation_date": null,
      "source": "https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card",
      "notes": null
    },
    {
      "id": "openai-20260903-fable51-terminal-bench-4-0",
      "model_slug": "claude-fable-5-1",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 4.0",
      "benchmark_version": null,
      "score_numeric": 55.8,
      "score_display": "55.8%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "GPT-6 Astra launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-09-03T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-6-astra/",
      "notes": "Cross-vendor comparison reported by OpenAI; preserve evaluator/harness when comparing."
    },
    {
      "id": "openai-20260903-fable51-terminal-bench-science-0-1",
      "model_slug": "claude-fable-5-1",
      "category": "math_reasoning",
      "benchmark_name": "Terminal-Bench Science 0.1",
      "benchmark_version": null,
      "score_numeric": 52.6,
      "score_display": "52.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "GPT-6 Astra launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-09-03T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-6-astra/",
      "notes": "Cross-vendor comparison reported by OpenAI; preserve evaluator/harness when comparing."
    },
    {
      "id": "tq-20260907-035-mythos5-70784b57a3dc2b028d4b",
      "model_slug": "claude-mythos-5",
      "category": "coding",
      "benchmark_name": "CursorBench",
      "benchmark_version": "3.2",
      "score_numeric": 70.5,
      "score_display": "70.5%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "max",
      "harness": "CursorBench 3.2",
      "evaluator": "SpaceXAI",
      "evaluation_date": "2026-08-12T00:00:00.000Z",
      "source": "https://x.ai/news/grok-4-6",
      "notes": "Shared underlying model with Claude Fable 5 per Anthropic; inherited ordinary capability evaluation. Cross-vendor comparison reported by SpaceXAI."
    },
    {
      "id": "tq-20260907-035-mythos5-2cdba935760303a5c485",
      "model_slug": "claude-mythos-5",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": null,
      "score_numeric": 70,
      "score_display": "70.0%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "max",
      "harness": "mini-swe-agent",
      "evaluator": "SpaceXAI",
      "evaluation_date": "2026-08-12T00:00:00.000Z",
      "source": "https://x.ai/news/grok-4-6",
      "notes": "Shared underlying model with Claude Fable 5 per Anthropic; inherited ordinary capability evaluation. Cross-vendor comparison reported by SpaceXAI."
    },
    {
      "id": "tq-20260907-035-mythos5-4220029289787d8ceddd",
      "model_slug": "claude-mythos-5",
      "category": "coding",
      "benchmark_name": "FrontierCode 1.1 Extended",
      "benchmark_version": null,
      "score_numeric": 63.6,
      "score_display": "63.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "max",
      "harness": null,
      "evaluator": "SpaceXAI",
      "evaluation_date": "2026-08-12T00:00:00.000Z",
      "source": "https://x.ai/news/grok-4-6",
      "notes": "Shared underlying model with Claude Fable 5 per Anthropic; inherited ordinary capability evaluation. Cross-vendor comparison reported by SpaceXAI."
    },
    {
      "id": "tq-20260907-035-mythos5-d0fe8d9cdfadb821885c",
      "model_slug": "claude-mythos-5",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA v2",
      "benchmark_version": null,
      "score_numeric": 1741,
      "score_display": "1741 Elo",
      "score_unit": "Elo",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "Artificial Analysis",
      "evaluator": "SpaceXAI",
      "evaluation_date": "2026-08-12T00:00:00.000Z",
      "source": "https://x.ai/news/grok-4-6",
      "notes": "Shared underlying model with Claude Fable 5 per Anthropic; inherited ordinary capability evaluation. Cross-vendor comparison; underlying evaluation is Artificial Analysis."
    },
    {
      "id": "tq-20260907-035-mythos5-b797f5090526c824bb4d",
      "model_slug": "claude-mythos-5",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": "Public",
      "score_numeric": 80.4,
      "score_display": "80.4%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "max",
      "harness": null,
      "evaluator": "SpaceXAI",
      "evaluation_date": "2026-07-16T00:00:00.000Z",
      "source": "https://x.ai/news/grok-4-5",
      "notes": "Shared underlying model with Claude Fable 5 per Anthropic; inherited ordinary capability evaluation. Cross-vendor comparison reported by SpaceXAI."
    },
    {
      "id": "tq-20260907-035-mythos5-66e13c193c59030581bf",
      "model_slug": "claude-mythos-5",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": null,
      "score_numeric": 84.3,
      "score_display": "84.3%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "max",
      "harness": "Terminus-2",
      "evaluator": "SpaceXAI",
      "evaluation_date": "2026-07-16T00:00:00.000Z",
      "source": "https://x.ai/news/grok-4-5",
      "notes": "Shared underlying model with Claude Fable 5 per Anthropic; inherited ordinary capability evaluation. Cross-vendor comparison reported by SpaceXAI."
    },
    {
      "id": "tq-20260907-035-mythos5-9e6cbfb4ee5c64c5f508",
      "model_slug": "claude-mythos-5",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 3.0",
      "benchmark_version": null,
      "score_numeric": 34.1,
      "score_display": "34.1%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "max",
      "harness": null,
      "evaluator": "SpaceXAI",
      "evaluation_date": "2026-08-12T00:00:00.000Z",
      "source": "https://x.ai/news/grok-4-6",
      "notes": "Shared underlying model with Claude Fable 5 per Anthropic; inherited ordinary capability evaluation. Cross-vendor comparison reported by SpaceXAI."
    },
    {
      "id": "tq-20260907-035-mythos5-196a5d0faf9757d34f06",
      "model_slug": "claude-mythos-5",
      "category": "coding",
      "benchmark_name": "Terminal-Bench Science 0.1",
      "benchmark_version": null,
      "score_numeric": 24.7,
      "score_display": "24.7%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Anthropic reproduction; Claude Code",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-09-01T00:00:00.000Z",
      "source": "https://www.anthropic.com/claude/fable",
      "notes": "Shared underlying model with Claude Fable 5 per Anthropic; inherited ordinary capability evaluation. Anthropic reproduction of the public leaderboard setup; public leaderboard score cited as 21.4%."
    },
    {
      "id": "tq-20260907-035-mythos51-d8f5fa3a4a8b4407f70e",
      "model_slug": "claude-mythos-5-1",
      "category": "coding",
      "benchmark_name": "CursorBench",
      "benchmark_version": "3.2.0",
      "score_numeric": 73.4,
      "score_display": "73.4",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Anthropic",
      "evaluation_date": null,
      "source": "https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card",
      "notes": "Shared underlying model with Claude Fable 5.1 per Anthropic; inherited ordinary capability evaluation."
    },
    {
      "id": "tq-20260907-035-mythos51-5a8468bd871159575d21",
      "model_slug": "claude-mythos-5-1",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": null,
      "score_numeric": 67.4,
      "score_display": "67.4%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "GPT-6 Astra launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-09-03T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-6-astra/",
      "notes": "Shared underlying model with Claude Fable 5.1 per Anthropic; inherited ordinary capability evaluation. Cross-vendor comparison reported by OpenAI; preserve evaluator/harness when comparing."
    },
    {
      "id": "tq-20260907-035-mythos51-fac46cb3863e465b0d02",
      "model_slug": "claude-mythos-5-1",
      "category": "coding",
      "benchmark_name": "FrontierCode 1.1 Extended",
      "benchmark_version": null,
      "score_numeric": 63.6,
      "score_display": "63.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "GPT-6 Astra launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-09-03T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-6-astra/",
      "notes": "Shared underlying model with Claude Fable 5.1 per Anthropic; inherited ordinary capability evaluation. Cross-vendor comparison reported by OpenAI; preserve evaluator/harness when comparing."
    },
    {
      "id": "tq-20260907-035-mythos51-6d01a6a7f6ac878bd241",
      "model_slug": "claude-mythos-5-1",
      "category": "coding",
      "benchmark_name": "FrontierCode 1.1 Main",
      "benchmark_version": null,
      "score_numeric": 50.9,
      "score_display": "50.9%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "GPT-6 Astra launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-09-03T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-6-astra/",
      "notes": "Shared underlying model with Claude Fable 5.1 per Anthropic; inherited ordinary capability evaluation. Cross-vendor comparison reported by OpenAI; preserve evaluator/harness when comparing."
    },
    {
      "id": "tq-20260907-035-mythos51-2800ecc436d1116c562a",
      "model_slug": "claude-mythos-5-1",
      "category": "math_reasoning",
      "benchmark_name": "FrontierMath Tier 4 (v2)",
      "benchmark_version": null,
      "score_numeric": 87.8,
      "score_display": "87.8%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "GPT-6 Astra launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-09-03T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-6-astra/",
      "notes": "Shared underlying model with Claude Fable 5.1 per Anthropic; inherited ordinary capability evaluation. Cross-vendor comparison reported by OpenAI; preserve evaluator/harness when comparing."
    },
    {
      "id": "tq-20260907-035-mythos51-3c2bc21f7f991326a910",
      "model_slug": "claude-mythos-5-1",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 93.7,
      "score_display": "93.7%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "GPT-6 Astra launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-09-03T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-6-astra/",
      "notes": "Shared underlying model with Claude Fable 5.1 per Anthropic; inherited ordinary capability evaluation. Cross-vendor comparison reported by OpenAI; preserve evaluator/harness when comparing."
    },
    {
      "id": "tq-20260907-035-mythos51-6bd95a9d1a17dfadec24",
      "model_slug": "claude-mythos-5-1",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 65,
      "score_display": "65.0%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": "GPT-6 Astra launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-09-03T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-6-astra/",
      "notes": "Shared underlying model with Claude Fable 5.1 per Anthropic; inherited ordinary capability evaluation. Cross-vendor comparison reported by OpenAI; preserve evaluator/harness when comparing."
    },
    {
      "id": "tq-20260907-035-mythos51-8b2b02e48717e2a49aef",
      "model_slug": "claude-mythos-5-1",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 65,
      "score_display": "65.0",
      "score_unit": "%",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Anthropic",
      "evaluation_date": null,
      "source": "https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card",
      "notes": "Shared underlying model with Claude Fable 5.1 per Anthropic; inherited ordinary capability evaluation."
    },
    {
      "id": "tq-20260907-035-mythos51-ee49bee8c2baeab98d24",
      "model_slug": "claude-mythos-5-1",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 60.9,
      "score_display": "60.9",
      "score_unit": "%",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Anthropic",
      "evaluation_date": null,
      "source": "https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card",
      "notes": "Shared underlying model with Claude Fable 5.1 per Anthropic; inherited ordinary capability evaluation."
    },
    {
      "id": "claude-mythos-5-1-terminal-bench-4-0-60-9-none-none-none-anthropic",
      "model_slug": "claude-mythos-5-1",
      "category": "coding",
      "benchmark_name": "Terminal-Bench",
      "benchmark_version": "4.0",
      "score_numeric": 60.9,
      "score_display": "60.9",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Anthropic",
      "evaluation_date": null,
      "source": "https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card",
      "notes": "Mythos has different safeguards from generally available Fable 5.1; do not compare as identical deployed systems."
    },
    {
      "id": "tq-20260907-035-mythos51-2070bffcef02b27c43af",
      "model_slug": "claude-mythos-5-1",
      "category": "coding",
      "benchmark_name": "Terminal-Bench",
      "benchmark_version": "4.0",
      "score_numeric": 55.8,
      "score_display": "55.8",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Anthropic",
      "evaluation_date": null,
      "source": "https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card",
      "notes": "Shared underlying model with Claude Fable 5.1 per Anthropic; inherited ordinary capability evaluation."
    },
    {
      "id": "tq-20260907-035-mythos51-7ef03a5c697b1517472c",
      "model_slug": "claude-mythos-5-1",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 4.0",
      "benchmark_version": null,
      "score_numeric": 55.8,
      "score_display": "55.8%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "GPT-6 Astra launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-09-03T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-6-astra/",
      "notes": "Shared underlying model with Claude Fable 5.1 per Anthropic; inherited ordinary capability evaluation. Cross-vendor comparison reported by OpenAI; preserve evaluator/harness when comparing."
    },
    {
      "id": "tq-20260907-035-mythos51-fca50d4581ef1a4b35af",
      "model_slug": "claude-mythos-5-1",
      "category": "math_reasoning",
      "benchmark_name": "Terminal-Bench Science 0.1",
      "benchmark_version": null,
      "score_numeric": 52.6,
      "score_display": "52.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "GPT-6 Astra launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-09-03T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-6-astra/",
      "notes": "Shared underlying model with Claude Fable 5.1 per Anthropic; inherited ordinary capability evaluation. Cross-vendor comparison reported by OpenAI; preserve evaluator/harness when comparing."
    },
    {
      "id": "tq-20260907-035-preview-81249a1a6b8307717c21",
      "model_slug": "claude-mythos-preview",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": null,
      "score_numeric": 86.9,
      "score_display": "86.9%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "Anthropic agentic search evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-04-07T00:00:00.000Z",
      "source": "https://www.anthropic.com/glasswing",
      "notes": "Anthropic reports Mythos Preview used 4.9× fewer tokens than Opus 4.6 in this evaluation."
    },
    {
      "id": "tq-20260907-035-preview-77536e080ff8d51cc994",
      "model_slug": "claude-mythos-preview",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 94.6,
      "score_display": "94.6%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "max",
      "harness": "Anthropic system-card evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-04-07T00:00:00.000Z",
      "source": "https://www.anthropic.com/glasswing",
      "notes": "System card gives 94.55%, rounded to one decimal for display consistency."
    },
    {
      "id": "tq-20260907-035-preview-e7a24ee598c7f6b73363",
      "model_slug": "claude-mythos-preview",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 64.7,
      "score_display": "64.7%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "Anthropic system-card evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-04-07T00:00:00.000Z",
      "source": "https://www.anthropic.com/glasswing",
      "notes": "With-tools condition."
    },
    {
      "id": "tq-20260907-035-preview-f5c82c1100ec9840694d",
      "model_slug": "claude-mythos-preview",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 56.8,
      "score_display": "56.8%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "max",
      "harness": "Anthropic system-card evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-04-07T00:00:00.000Z",
      "source": "https://www.anthropic.com/glasswing",
      "notes": "No-tools condition."
    },
    {
      "id": "tq-20260907-035-preview-4327a15f046d26055276",
      "model_slug": "claude-mythos-preview",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld-Verified",
      "benchmark_version": null,
      "score_numeric": 79.6,
      "score_display": "79.6%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "Anthropic computer-use evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-04-07T00:00:00.000Z",
      "source": "https://www.anthropic.com/glasswing",
      "notes": "Direct Mythos Preview evaluation; do not transfer this value to safeguarded Fable/Mythos descendants."
    },
    {
      "id": "tq-20260907-035-preview-e8a29ffb50703dabe7fa",
      "model_slug": "claude-mythos-preview",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": null,
      "score_numeric": 77.8,
      "score_display": "77.8%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "max",
      "harness": "Anthropic Mythos Preview system-card harness",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-04-07T00:00:00.000Z",
      "source": "https://www.anthropic.com/glasswing",
      "notes": "Averaged over five trials."
    },
    {
      "id": "tq-20260907-035-preview-6ad59c8c00cdad082143",
      "model_slug": "claude-mythos-preview",
      "category": "coding",
      "benchmark_name": "SWE-bench Verified",
      "benchmark_version": null,
      "score_numeric": 93.9,
      "score_display": "93.9%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "max",
      "harness": "Anthropic Mythos Preview system-card harness",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-04-07T00:00:00.000Z",
      "source": "https://www.anthropic.com/glasswing",
      "notes": "Averaged over five trials."
    },
    {
      "id": "tq-20260907-035-preview-cc717e70e935a06b5471",
      "model_slug": "claude-mythos-preview",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.0",
      "benchmark_version": null,
      "score_numeric": 82,
      "score_display": "82.0%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "max",
      "harness": "Terminus-2; 1M-token task budget",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-04-07T00:00:00.000Z",
      "source": "https://www.anthropic.com/glasswing",
      "notes": "1× guaranteed / 3× ceiling resources, averaged over five attempts per task."
    },
    {
      "id": "tq-20260907-035-preview-335477daedea8c3126cd",
      "model_slug": "claude-mythos-preview",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": "4h timeout",
      "score_numeric": 92.1,
      "score_display": "92.1%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "max",
      "harness": "Terminus-2; Terminal-Bench 2.1 updates",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-04-07T00:00:00.000Z",
      "source": "https://www.anthropic.com/glasswing",
      "notes": "Anthropic reports 92.1% after increasing timeout limits to four hours and applying Terminal-Bench 2.1 updates."
    },
    {
      "id": "anthropic-20260416-opus46-browsecomp",
      "model_slug": "claude-opus-4-6",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": null,
      "score_numeric": 83.7,
      "score_display": "83.7%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Anthropic comparison evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-04-16T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-opus-4-7",
      "notes": null
    },
    {
      "id": "aa-20260205-opus46-gdpval",
      "model_slug": "claude-opus-4-6",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA",
      "benchmark_version": null,
      "score_numeric": 1606,
      "score_display": "1606 Elo",
      "score_unit": "Elo",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Artificial Analysis",
      "evaluator": "Artificial Analysis",
      "evaluation_date": "2026-02-05T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-opus-4-6",
      "notes": "Original GDPval-AA benchmark; do not compare as if it were GDPval-AA v2."
    },
    {
      "id": "anthropic-20260416-opus46-gpqa",
      "model_slug": "claude-opus-4-6",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 91.3,
      "score_display": "91.3%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "high",
      "harness": "Anthropic comparison evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-04-16T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-opus-4-7",
      "notes": null
    },
    {
      "id": "anthropic-20260416-opus46-hle-no-tools",
      "model_slug": "claude-opus-4-6",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 40,
      "score_display": "40.0%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "high",
      "harness": "Anthropic comparison evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-04-16T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-opus-4-7",
      "notes": null
    },
    {
      "id": "anthropic-20260416-opus46-hle-tools",
      "model_slug": "claude-opus-4-6",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 53.3,
      "score_display": "53.3%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Search + fetch + code + compaction",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-04-16T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-opus-4-7",
      "notes": null
    },
    {
      "id": "anthropic-20260416-opus46-mcp-atlas",
      "model_slug": "claude-opus-4-6",
      "category": "agentic_computer_use",
      "benchmark_name": "MCP Atlas",
      "benchmark_version": null,
      "score_numeric": 75.8,
      "score_display": "75.8%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Updated Anthropic grading",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-04-16T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-opus-4-7",
      "notes": "Uses the updated grading reported with the Opus 4.7 comparison table."
    },
    {
      "id": "anthropic-20260416-opus46-osworld-verified",
      "model_slug": "claude-opus-4-6",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld-Verified",
      "benchmark_version": null,
      "score_numeric": 72.7,
      "score_display": "72.7%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Anthropic comparison evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-04-16T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-opus-4-7",
      "notes": null
    },
    {
      "id": "anthropic-20260416-opus46-swe-pro",
      "model_slug": "claude-opus-4-6",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": null,
      "score_numeric": 53.4,
      "score_display": "53.4%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "high",
      "harness": "Anthropic comparison evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-04-16T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-opus-4-7",
      "notes": null
    },
    {
      "id": "anthropic-20260416-opus46-swe-verified",
      "model_slug": "claude-opus-4-6",
      "category": "coding",
      "benchmark_name": "SWE-bench Verified",
      "benchmark_version": "Verified",
      "score_numeric": 80.8,
      "score_display": "80.8%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "high",
      "harness": "Anthropic comparison evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-04-16T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-opus-4-7",
      "notes": null
    },
    {
      "id": "anthropic-20260416-opus46-terminal20",
      "model_slug": "claude-opus-4-6",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.0",
      "benchmark_version": null,
      "score_numeric": 65.4,
      "score_display": "65.4%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "thinking disabled",
      "harness": "Terminus-2",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-04-16T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-opus-4-7",
      "notes": null
    },
    {
      "id": "anthropic-20260416-opus47-browsecomp",
      "model_slug": "claude-opus-4-7",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": null,
      "score_numeric": 79.3,
      "score_display": "79.3%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Anthropic launch evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-04-16T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-opus-4-7",
      "notes": null
    },
    {
      "id": "anthropic-20260416-opus47-gpqa",
      "model_slug": "claude-opus-4-7",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 94.2,
      "score_display": "94.2%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "high",
      "harness": "Anthropic launch evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-04-16T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-opus-4-7",
      "notes": null
    },
    {
      "id": "anthropic-20260416-opus47-hle-no-tools",
      "model_slug": "claude-opus-4-7",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 46.9,
      "score_display": "46.9%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "high",
      "harness": "Anthropic launch evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-04-16T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-opus-4-7",
      "notes": null
    },
    {
      "id": "anthropic-20260416-opus47-hle-tools",
      "model_slug": "claude-opus-4-7",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 54.7,
      "score_display": "54.7%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Anthropic launch evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-04-16T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-opus-4-7",
      "notes": null
    },
    {
      "id": "anthropic-20260416-opus47-mcp-atlas",
      "model_slug": "claude-opus-4-7",
      "category": "agentic_computer_use",
      "benchmark_name": "MCP Atlas",
      "benchmark_version": null,
      "score_numeric": 77.3,
      "score_display": "77.3%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Updated Anthropic grading",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-04-16T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-opus-4-7",
      "notes": null
    },
    {
      "id": "anthropic-20260528-opus47-osworld-verified",
      "model_slug": "claude-opus-4-7",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld-Verified",
      "benchmark_version": null,
      "score_numeric": 82.3,
      "score_display": "82.3%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Revised Anthropic methodology",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-05-28T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-opus-4-8",
      "notes": "Updated from the original Opus 4.7 launch value using Anthropic's revised OSWorld methodology."
    },
    {
      "id": "anthropic-20260416-opus47-swe-pro",
      "model_slug": "claude-opus-4-7",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": null,
      "score_numeric": 64.3,
      "score_display": "64.3%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "high",
      "harness": "Anthropic launch evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-04-16T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-opus-4-7",
      "notes": null
    },
    {
      "id": "anthropic-20260416-opus47-swe-verified",
      "model_slug": "claude-opus-4-7",
      "category": "coding",
      "benchmark_name": "SWE-bench Verified",
      "benchmark_version": "Verified",
      "score_numeric": 87.6,
      "score_display": "87.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "high",
      "harness": "Anthropic launch evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-04-16T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-opus-4-7",
      "notes": null
    },
    {
      "id": "anthropic-20260416-opus47-terminal20",
      "model_slug": "claude-opus-4-7",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.0",
      "benchmark_version": null,
      "score_numeric": 69.4,
      "score_display": "69.4%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "thinking disabled",
      "harness": "Terminus-2",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-04-16T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-opus-4-7",
      "notes": "Anthropic reports this run with thinking disabled."
    },
    {
      "id": "aa-20260630-opus48-gdpval-v2",
      "model_slug": "claude-opus-4-8",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA v2",
      "benchmark_version": null,
      "score_numeric": 1615,
      "score_display": "1615 Elo",
      "score_unit": "Elo",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Artificial Analysis",
      "evaluator": "Artificial Analysis",
      "evaluation_date": "2026-06-30T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-sonnet-5",
      "notes": "GDPval-AA v2."
    },
    {
      "id": "anthropic-20260630-opus48-gpqa",
      "model_slug": "claude-opus-4-8",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 93.6,
      "score_display": "93.6%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "high",
      "harness": "Anthropic Sonnet 5 comparison table",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-06-30T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-sonnet-5",
      "notes": null
    },
    {
      "id": "anthropic-20260630-opus48-hle-no-tools",
      "model_slug": "claude-opus-4-8",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 49.8,
      "score_display": "49.8%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "high",
      "harness": "Anthropic Sonnet 5 comparison table",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-06-30T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-sonnet-5",
      "notes": null
    },
    {
      "id": "anthropic-20260630-opus48-hle-tools",
      "model_slug": "claude-opus-4-8",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 57.9,
      "score_display": "57.9%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Anthropic Sonnet 5 comparison table",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-06-30T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-sonnet-5",
      "notes": null
    },
    {
      "id": "anthropic-20260630-opus48-osworld-verified",
      "model_slug": "claude-opus-4-8",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld-Verified",
      "benchmark_version": null,
      "score_numeric": 83.4,
      "score_display": "83.4%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Anthropic Sonnet 5 comparison table",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-06-30T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-sonnet-5",
      "notes": null
    },
    {
      "id": "anthropic-20260528-opus48-swe-pro",
      "model_slug": "claude-opus-4-8",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": null,
      "score_numeric": 69.2,
      "score_display": "69.2%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "high",
      "harness": "Anthropic comparison evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-05-28T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-opus-4-8",
      "notes": null
    },
    {
      "id": "anthropic-20260528-opus48-swe-verified",
      "model_slug": "claude-opus-4-8",
      "category": "coding",
      "benchmark_name": "SWE-bench Verified",
      "benchmark_version": "Verified",
      "score_numeric": 88.6,
      "score_display": "88.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "high",
      "harness": "Anthropic comparison evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-05-28T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-opus-4-8",
      "notes": null
    },
    {
      "id": "anthropic-20260630-opus48-terminal21",
      "model_slug": "claude-opus-4-8",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": null,
      "score_numeric": 82.7,
      "score_display": "82.7%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "high",
      "harness": "Terminus-2 public harness",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-06-30T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-sonnet-5",
      "notes": null
    },
    {
      "id": "anthropic-20260724-opus5-arcagi3",
      "model_slug": "claude-opus-5",
      "category": "math_reasoning",
      "benchmark_name": "ARC-AGI",
      "benchmark_version": "3",
      "score_numeric": 30.2,
      "score_display": "30.2%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "high",
      "harness": "ARC-AGI-3 evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-07-24T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-opus-5",
      "notes": null
    },
    {
      "id": "anthropic-20260724-opus5-automation",
      "model_slug": "claude-opus-5",
      "category": "agentic_computer_use",
      "benchmark_name": "AutomationBench",
      "benchmark_version": null,
      "score_numeric": 26,
      "score_display": "26.0%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Anthropic launch evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-07-24T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-opus-5",
      "notes": null
    },
    {
      "id": "anthropic-20260724-opus5-browsecomp",
      "model_slug": "claude-opus-5",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": null,
      "score_numeric": 90.8,
      "score_display": "90.8%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Anthropic launch evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-07-24T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-opus-5",
      "notes": null
    },
    {
      "id": "aa-20260724-opus5-gdpval-v2",
      "model_slug": "claude-opus-5",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA v2",
      "benchmark_version": null,
      "score_numeric": 1861,
      "score_display": "1861 Elo",
      "score_unit": "Elo",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Artificial Analysis",
      "evaluator": "Artificial Analysis",
      "evaluation_date": "2026-07-24T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-opus-5",
      "notes": "GDPval-AA v2; do not conflate with the original GDPval-AA benchmark."
    },
    {
      "id": "anthropic-20260724-opus5-hle-no-tools",
      "model_slug": "claude-opus-5",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 56.3,
      "score_display": "56.3%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "high",
      "harness": "Anthropic launch evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-07-24T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-opus-5",
      "notes": null
    },
    {
      "id": "anthropic-20260724-opus5-hle-tools",
      "model_slug": "claude-opus-5",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 64.7,
      "score_display": "64.7%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Search + code tools",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-07-24T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-opus-5",
      "notes": null
    },
    {
      "id": "anthropic-20260724-opus5-mcp-atlas",
      "model_slug": "claude-opus-5",
      "category": "agentic_computer_use",
      "benchmark_name": "MCP Atlas",
      "benchmark_version": null,
      "score_numeric": 85.8,
      "score_display": "85.8%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Anthropic launch evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-07-24T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-opus-5",
      "notes": null
    },
    {
      "id": "anthropic-20260724-opus5-osworld20-first",
      "model_slug": "claude-opus-5",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld 2.0",
      "benchmark_version": "First-attempt success",
      "score_numeric": 70.57,
      "score_display": "70.57%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Anthropic system-card evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-07-24T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-opus-5",
      "notes": "First-attempt success rate; keep distinct from partial-credit and other OSWorld 2.0 methodologies."
    },
    {
      "id": "anthropic-20260724-opus5-swe-pro",
      "model_slug": "claude-opus-5",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": null,
      "score_numeric": 79.2,
      "score_display": "79.2%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "high",
      "harness": "Anthropic launch evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-07-24T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-opus-5",
      "notes": null
    },
    {
      "id": "anthropic-20260724-opus5-swe-verified",
      "model_slug": "claude-opus-5",
      "category": "coding",
      "benchmark_name": "SWE-bench Verified",
      "benchmark_version": "Verified",
      "score_numeric": 96,
      "score_display": "96.0%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "high",
      "harness": "Anthropic launch evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-07-24T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-opus-5",
      "notes": null
    },
    {
      "id": "anthropic-20260217-sonnet46-arcagi2",
      "model_slug": "claude-sonnet-4-6",
      "category": "math_reasoning",
      "benchmark_name": "ARC-AGI",
      "benchmark_version": "2",
      "score_numeric": 58.3,
      "score_display": "58.3%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "high",
      "harness": "ARC-AGI-2 evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-02-17T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-sonnet-4-6",
      "notes": null
    },
    {
      "id": "anthropic-20260630-sonnet46-browsecomp",
      "model_slug": "claude-sonnet-4-6",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": "Single-agent",
      "score_numeric": 74.01,
      "score_display": "74.01%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Anthropic system-card revised value",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-06-30T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-sonnet-5",
      "notes": "Uses the revised single-agent score rather than the earlier launch-table value."
    },
    {
      "id": "aa-20260630-sonnet46-gdpval-v2",
      "model_slug": "claude-sonnet-4-6",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA v2",
      "benchmark_version": null,
      "score_numeric": 1395,
      "score_display": "1395 Elo",
      "score_unit": "Elo",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Artificial Analysis",
      "evaluator": "Artificial Analysis",
      "evaluation_date": "2026-06-30T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-sonnet-5",
      "notes": "GDPval-AA v2."
    },
    {
      "id": "anthropic-20260630-sonnet46-hle-no-tools",
      "model_slug": "claude-sonnet-4-6",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 34.6,
      "score_display": "34.6%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "high",
      "harness": "Updated Anthropic grader",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-06-30T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-sonnet-5",
      "notes": "Updated grader value from Anthropic's Sonnet 5 comparison table."
    },
    {
      "id": "anthropic-20260630-sonnet46-hle-tools",
      "model_slug": "claude-sonnet-4-6",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 46.8,
      "score_display": "46.8%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Updated Anthropic grader",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-06-30T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-sonnet-5",
      "notes": "Updated grader value from Anthropic's Sonnet 5 comparison table."
    },
    {
      "id": "anthropic-20260630-sonnet46-osworld-verified",
      "model_slug": "claude-sonnet-4-6",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld-Verified",
      "benchmark_version": null,
      "score_numeric": 78.5,
      "score_display": "78.5%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Updated Anthropic methodology",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-06-30T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-sonnet-5",
      "notes": null
    },
    {
      "id": "anthropic-20260630-sonnet46-swe-pro",
      "model_slug": "claude-sonnet-4-6",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": null,
      "score_numeric": 58.1,
      "score_display": "58.1%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "high",
      "harness": "Anthropic Sonnet 5 comparison table",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-06-30T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-sonnet-5",
      "notes": null
    },
    {
      "id": "anthropic-20260217-sonnet46-swe-verified",
      "model_slug": "claude-sonnet-4-6",
      "category": "coding",
      "benchmark_name": "SWE-bench Verified",
      "benchmark_version": "Verified",
      "score_numeric": 79.6,
      "score_display": "79.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "high",
      "harness": "Anthropic launch evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-02-17T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-sonnet-4-6",
      "notes": null
    },
    {
      "id": "anthropic-20260217-sonnet46-terminal20",
      "model_slug": "claude-sonnet-4-6",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.0",
      "benchmark_version": null,
      "score_numeric": 59.1,
      "score_display": "59.1%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "thinking disabled",
      "harness": "Terminus-2",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-02-17T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-sonnet-4-6",
      "notes": "Anthropic reports this run with thinking disabled."
    },
    {
      "id": "anthropic-20260630-sonnet46-terminal21",
      "model_slug": "claude-sonnet-4-6",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": null,
      "score_numeric": 67,
      "score_display": "67.0%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "high",
      "harness": "Terminus-2 public harness",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-06-30T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-sonnet-5",
      "notes": "Later Terminal-Bench 2.1 run; kept distinct from the original Terminal-Bench 2.0 result."
    },
    {
      "id": "anthropic-20260630-sonnet5-automation",
      "model_slug": "claude-sonnet-5",
      "category": "agentic_computer_use",
      "benchmark_name": "AutomationBench",
      "benchmark_version": null,
      "score_numeric": 13.5,
      "score_display": "13.5%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Anthropic launch evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-06-30T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-sonnet-5",
      "notes": null
    },
    {
      "id": "anthropic-20260630-sonnet5-browsecomp",
      "model_slug": "claude-sonnet-5",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": "Single-agent",
      "score_numeric": 84.7,
      "score_display": "84.7%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Anthropic launch evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-06-30T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-sonnet-5",
      "notes": null
    },
    {
      "id": "anthropic-20260630-sonnet5-frontiercode-main",
      "model_slug": "claude-sonnet-5",
      "category": "coding",
      "benchmark_name": "FrontierCode 1.1 Main",
      "benchmark_version": null,
      "score_numeric": 38.8,
      "score_display": "38.8%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "high",
      "harness": "Anthropic launch evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-06-30T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-sonnet-5",
      "notes": null
    },
    {
      "id": "aa-20260630-sonnet5-gdpval-v2",
      "model_slug": "claude-sonnet-5",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA v2",
      "benchmark_version": null,
      "score_numeric": 1618,
      "score_display": "1618 Elo",
      "score_unit": "Elo",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Artificial Analysis",
      "evaluator": "Artificial Analysis",
      "evaluation_date": "2026-06-30T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-sonnet-5",
      "notes": "GDPval-AA v2."
    },
    {
      "id": "anthropic-20260630-sonnet5-hle-no-tools",
      "model_slug": "claude-sonnet-5",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 43.2,
      "score_display": "43.2%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "high",
      "harness": "Anthropic launch evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-06-30T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-sonnet-5",
      "notes": null
    },
    {
      "id": "anthropic-20260630-sonnet5-hle-tools",
      "model_slug": "claude-sonnet-5",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 57.4,
      "score_display": "57.4%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Anthropic launch evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-06-30T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-sonnet-5",
      "notes": null
    },
    {
      "id": "anthropic-20260630-sonnet5-osworld-verified",
      "model_slug": "claude-sonnet-5",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld-Verified",
      "benchmark_version": null,
      "score_numeric": 81.2,
      "score_display": "81.2%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Anthropic launch evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-06-30T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-sonnet-5",
      "notes": null
    },
    {
      "id": "anthropic-20260630-sonnet5-swe-pro",
      "model_slug": "claude-sonnet-5",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": null,
      "score_numeric": 63.2,
      "score_display": "63.2%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "high",
      "harness": "Anthropic launch evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-06-30T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-sonnet-5",
      "notes": null
    },
    {
      "id": "anthropic-20260630-sonnet5-swe-verified",
      "model_slug": "claude-sonnet-5",
      "category": "coding",
      "benchmark_name": "SWE-bench Verified",
      "benchmark_version": "Verified",
      "score_numeric": 85.2,
      "score_display": "85.2%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "high",
      "harness": "Anthropic launch evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-06-30T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-sonnet-5",
      "notes": null
    },
    {
      "id": "anthropic-20260630-sonnet5-terminal21",
      "model_slug": "claude-sonnet-5",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": null,
      "score_numeric": 80.4,
      "score_display": "80.4%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "high",
      "harness": "Terminus-2 public harness",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-06-30T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-sonnet-5",
      "notes": null
    },
    {
      "id": "anthropic-20260630-sonnet5-toolathlon",
      "model_slug": "claude-sonnet-5",
      "category": "agentic_computer_use",
      "benchmark_name": "Toolathlon",
      "benchmark_version": null,
      "score_numeric": 54.3,
      "score_display": "54.3%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Anthropic launch evaluation",
      "evaluator": "Anthropic",
      "evaluation_date": "2026-06-30T00:00:00.000Z",
      "source": "https://www.anthropic.com/news/claude-sonnet-5",
      "notes": null
    },
    {
      "id": "tq-20260907-036-a88598377915c235a00b",
      "model_slug": "command-a-plus",
      "category": "multimodal",
      "benchmark_name": "CharXiv",
      "benchmark_version": "reasoning",
      "score_numeric": 52.7,
      "score_display": "52.7%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": "Cohere Command A+ launch evaluation",
      "evaluator": "Cohere",
      "evaluation_date": "2026-05-20T00:00:00.000Z",
      "source": "https://cohere.com/blog/command-a-plus",
      "notes": null
    },
    {
      "id": "tq-20260907-036-19ff36a1cc3a8deaf7c7",
      "model_slug": "command-a-plus",
      "category": "multimodal",
      "benchmark_name": "MMMU",
      "benchmark_version": null,
      "score_numeric": 75.1,
      "score_display": "75.1%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": "Cohere Command A+ launch evaluation",
      "evaluator": "Cohere",
      "evaluation_date": "2026-05-20T00:00:00.000Z",
      "source": "https://cohere.com/blog/command-a-plus",
      "notes": null
    },
    {
      "id": "tq-20260907-036-84c746d15e3163fa3906",
      "model_slug": "command-a-plus",
      "category": "multimodal",
      "benchmark_name": "MMMU-Pro",
      "benchmark_version": null,
      "score_numeric": 63,
      "score_display": "63%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": "Cohere Command A+ launch evaluation",
      "evaluator": "Cohere",
      "evaluation_date": "2026-05-20T00:00:00.000Z",
      "source": "https://cohere.com/blog/command-a-plus",
      "notes": null
    },
    {
      "id": "tq-20260907-036-ebaf39b3035aa2e5484d",
      "model_slug": "command-a-plus",
      "category": "multimodal",
      "benchmark_name": "MathVista",
      "benchmark_version": null,
      "score_numeric": 80.6,
      "score_display": "80.6%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": "Cohere Command A+ launch evaluation",
      "evaluator": "Cohere",
      "evaluation_date": "2026-05-20T00:00:00.000Z",
      "source": "https://cohere.com/blog/command-a-plus",
      "notes": null
    },
    {
      "id": "command-a-plus-terminal-bench-hard-25-none-none-none-cohere",
      "model_slug": "command-a-plus",
      "category": "coding",
      "benchmark_name": "Terminal-Bench",
      "benchmark_version": "Hard",
      "score_numeric": 25,
      "score_display": "25",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Cohere",
      "evaluation_date": null,
      "source": "https://cohere.com/blog/command-a-plus",
      "notes": null
    },
    {
      "id": "tq-20260907-036-cae9802dd9de50c6c482",
      "model_slug": "command-a-plus",
      "category": "coding",
      "benchmark_name": "Terminal-Bench",
      "benchmark_version": "Hard",
      "score_numeric": 25,
      "score_display": "25%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Cohere Command A+ launch evaluation",
      "evaluator": "Cohere",
      "evaluation_date": "2026-05-20T00:00:00.000Z",
      "source": "https://cohere.com/blog/command-a-plus",
      "notes": "Cohere launch result; improved from 3% for Command A Reasoning."
    },
    {
      "id": "command-a-plus-τ²-telecom-85-none-none-none-cohere",
      "model_slug": "command-a-plus",
      "category": "agentic_computer_use",
      "benchmark_name": "τ² Telecom",
      "benchmark_version": null,
      "score_numeric": 85,
      "score_display": "85",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Cohere",
      "evaluation_date": null,
      "source": "https://cohere.com/blog/command-a-plus",
      "notes": null
    },
    {
      "id": "tq-20260907-036-5f1d651d40485b8272e2",
      "model_slug": "command-a-plus",
      "category": "agentic_computer_use",
      "benchmark_name": "τ²-bench Telecom",
      "benchmark_version": null,
      "score_numeric": 85,
      "score_display": "85%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Cohere Command A+ launch evaluation",
      "evaluator": "Cohere",
      "evaluation_date": "2026-05-20T00:00:00.000Z",
      "source": "https://cohere.com/blog/command-a-plus",
      "notes": "Cohere launch result; improved from 37% for Command A Reasoning."
    },
    {
      "id": "tq-20260907-dsv4flash-ale",
      "model_slug": "deepseek-v4-flash",
      "category": "agentic_computer_use",
      "benchmark_name": "Agents' Last Exam",
      "benchmark_version": null,
      "score_numeric": 25.2,
      "score_display": "25.2%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "DeepSeek Harness minimal mode",
      "evaluator": "DeepSeek",
      "evaluation_date": "2026-07-31T00:00:00.000Z",
      "source": "https://api-docs.deepseek.com/updates",
      "notes": null
    },
    {
      "id": "tq-20260907-dsv4flash-automation",
      "model_slug": "deepseek-v4-flash",
      "category": "agentic_computer_use",
      "benchmark_name": "AutomationBench",
      "benchmark_version": "Public",
      "score_numeric": 25.1,
      "score_display": "25.1%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "DeepSeek Harness minimal mode",
      "evaluator": "DeepSeek",
      "evaluation_date": "2026-07-31T00:00:00.000Z",
      "source": "https://api-docs.deepseek.com/updates",
      "notes": null
    },
    {
      "id": "tq-20260907-dsv4flash-deepswe",
      "model_slug": "deepseek-v4-flash",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": null,
      "score_numeric": 54.4,
      "score_display": "54.4%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "max",
      "harness": "DeepSeek Harness minimal mode",
      "evaluator": "DeepSeek",
      "evaluation_date": "2026-07-31T00:00:00.000Z",
      "source": "https://api-docs.deepseek.com/updates",
      "notes": null
    },
    {
      "id": "tq-20260907-dsv4flash-hle-no-tools",
      "model_slug": "deepseek-v4-flash",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 37.8,
      "score_display": "37.8%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "max",
      "harness": "DeepSeek V4 Pro comparison evaluation",
      "evaluator": "DeepSeek",
      "evaluation_date": "2026-08-13T00:00:00.000Z",
      "source": "https://api-docs.deepseek.com/updates",
      "notes": null
    },
    {
      "id": "tq-20260907-dsv4flash-hle-tools",
      "model_slug": "deepseek-v4-flash",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 51.5,
      "score_display": "51.5%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "DeepSeek V4 Pro comparison evaluation",
      "evaluator": "DeepSeek",
      "evaluation_date": "2026-08-13T00:00:00.000Z",
      "source": "https://api-docs.deepseek.com/updates",
      "notes": null
    },
    {
      "id": "deepseek-v4-flash-terminal-bench-2-1-82-7-none-max-deepseek-harness-minimal-mode-deepseek",
      "model_slug": "deepseek-v4-flash",
      "category": "coding",
      "benchmark_name": "Terminal-Bench",
      "benchmark_version": "2.1",
      "score_numeric": 82.7,
      "score_display": "82.7",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": "max",
      "harness": "DeepSeek Harness minimal mode",
      "evaluator": "DeepSeek",
      "evaluation_date": null,
      "source": "https://deepseek.com/en/news/v4-preview/",
      "notes": null
    },
    {
      "id": "tq-20260907-dsv4flash-terminal21",
      "model_slug": "deepseek-v4-flash",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": null,
      "score_numeric": 82.7,
      "score_display": "82.7%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "max",
      "harness": "DeepSeek Harness minimal mode",
      "evaluator": "DeepSeek",
      "evaluation_date": "2026-07-31T00:00:00.000Z",
      "source": "https://api-docs.deepseek.com/updates",
      "notes": "Public Code Agent tasks; top_p 0.95 and temperature 1.0 in the documented setup."
    },
    {
      "id": "deepseek-v4-flash-toolathlon-70-3-none-none-none-deepseek",
      "model_slug": "deepseek-v4-flash",
      "category": "agentic_computer_use",
      "benchmark_name": "Toolathlon",
      "benchmark_version": null,
      "score_numeric": 70.3,
      "score_display": "70.3",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "DeepSeek",
      "evaluation_date": null,
      "source": "https://deepseek.com/en/news/v4-preview/",
      "notes": null
    },
    {
      "id": "tq-20260907-dsv4flash-toolathlon",
      "model_slug": "deepseek-v4-flash",
      "category": "agentic_computer_use",
      "benchmark_name": "Toolathlon",
      "benchmark_version": "Verified",
      "score_numeric": 70.3,
      "score_display": "70.3%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "DeepSeek Harness minimal mode",
      "evaluator": "DeepSeek",
      "evaluation_date": "2026-07-31T00:00:00.000Z",
      "source": "https://api-docs.deepseek.com/updates",
      "notes": null
    },
    {
      "id": "tq-20260907-036-33f6421cebd23b8e8d9c",
      "model_slug": "deepseek-v4-flash-vision-exp",
      "category": "agentic_computer_use",
      "benchmark_name": "Agents' Last Exam",
      "benchmark_version": null,
      "score_numeric": 27.3,
      "score_display": "27.3%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "DeepSeek Harness minimal mode",
      "evaluator": "DeepSeek",
      "evaluation_date": "2026-08-21T00:00:00.000Z",
      "source": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp",
      "notes": "temperature=1.0; top_p=0.95 for DeepSeek text-agent evaluations."
    },
    {
      "id": "tq-20260907-036-c1eb2453741209ad553d",
      "model_slug": "deepseek-v4-flash-vision-exp",
      "category": "agentic_computer_use",
      "benchmark_name": "ApexBench",
      "benchmark_version": "Pass@1",
      "score_numeric": 36.5,
      "score_display": "36.5%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "DeepSeek Harness minimal mode",
      "evaluator": "DeepSeek",
      "evaluation_date": "2026-08-21T00:00:00.000Z",
      "source": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp",
      "notes": "temperature=1.0; top_p=0.95 for DeepSeek text-agent evaluations."
    },
    {
      "id": "tq-20260907-036-6f35ed734242d183f993",
      "model_slug": "deepseek-v4-flash-vision-exp",
      "category": "agentic_computer_use",
      "benchmark_name": "AutomationBench",
      "benchmark_version": "Public",
      "score_numeric": 25.7,
      "score_display": "25.7%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "DeepSeek Harness minimal mode",
      "evaluator": "DeepSeek",
      "evaluation_date": "2026-08-21T00:00:00.000Z",
      "source": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp",
      "notes": "temperature=1.0; top_p=0.95 for DeepSeek text-agent evaluations."
    },
    {
      "id": "tq-20260907-036-02bfa136cc27af52134c",
      "model_slug": "deepseek-v4-flash-vision-exp",
      "category": "multimodal",
      "benchmark_name": "Chartography",
      "benchmark_version": null,
      "score_numeric": 64.3,
      "score_display": "64.3%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "max",
      "harness": "DeepSeek V4 Flash Vision launch evaluation",
      "evaluator": "DeepSeek",
      "evaluation_date": "2026-08-21T00:00:00.000Z",
      "source": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp",
      "notes": null
    },
    {
      "id": "tq-20260907-036-67604ce3aca808f8bc35",
      "model_slug": "deepseek-v4-flash-vision-exp",
      "category": "professional",
      "benchmark_name": "DSBench-Hard",
      "benchmark_version": null,
      "score_numeric": 63.6,
      "score_display": "63.6%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "DeepSeek Harness minimal mode",
      "evaluator": "DeepSeek",
      "evaluation_date": "2026-08-21T00:00:00.000Z",
      "source": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp",
      "notes": "temperature=1.0; top_p=0.95 for DeepSeek text-agent evaluations."
    },
    {
      "id": "tq-20260907-036-b858bd5e40017471167f",
      "model_slug": "deepseek-v4-flash-vision-exp",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": null,
      "score_numeric": 59.3,
      "score_display": "59.3%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "max",
      "harness": "DeepSeek Harness minimal mode",
      "evaluator": "DeepSeek",
      "evaluation_date": "2026-08-21T00:00:00.000Z",
      "source": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp",
      "notes": "temperature=1.0; top_p=0.95 for DeepSeek text-agent evaluations."
    },
    {
      "id": "tq-20260907-036-aeddd4b05665f7ec1edc",
      "model_slug": "deepseek-v4-flash-vision-exp",
      "category": "coding",
      "benchmark_name": "NL2Repo",
      "benchmark_version": null,
      "score_numeric": 57.7,
      "score_display": "57.7%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "max",
      "harness": "DeepSeek Harness minimal mode",
      "evaluator": "DeepSeek",
      "evaluation_date": "2026-08-21T00:00:00.000Z",
      "source": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp",
      "notes": "temperature=1.0; top_p=0.95 for DeepSeek text-agent evaluations."
    },
    {
      "id": "tq-20260907-036-809b9390703639932802",
      "model_slug": "deepseek-v4-flash-vision-exp",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": null,
      "score_numeric": 83.9,
      "score_display": "83.9%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "max",
      "harness": "DeepSeek Harness minimal mode",
      "evaluator": "DeepSeek",
      "evaluation_date": "2026-08-21T00:00:00.000Z",
      "source": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp",
      "notes": "temperature=1.0; top_p=0.95 for DeepSeek text-agent evaluations."
    },
    {
      "id": "tq-20260907-036-8455fed18f2d55ddadec",
      "model_slug": "deepseek-v4-flash-vision-exp",
      "category": "agentic_computer_use",
      "benchmark_name": "Toolathlon",
      "benchmark_version": "Verified",
      "score_numeric": 75.9,
      "score_display": "75.9%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "DeepSeek Harness minimal mode",
      "evaluator": "DeepSeek",
      "evaluation_date": "2026-08-21T00:00:00.000Z",
      "source": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp",
      "notes": "temperature=1.0; top_p=0.95 for DeepSeek text-agent evaluations."
    },
    {
      "id": "tq-20260907-036-eff6c8ec3d53bfdf4df3",
      "model_slug": "deepseek-v4-flash-vision-exp",
      "category": "multimodal",
      "benchmark_name": "ZeroBench",
      "benchmark_version": "Pass@5",
      "score_numeric": 35,
      "score_display": "35%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "max",
      "harness": "DeepSeek V4 Flash Vision launch evaluation",
      "evaluator": "DeepSeek",
      "evaluation_date": "2026-08-21T00:00:00.000Z",
      "source": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp",
      "notes": null
    },
    {
      "id": "tq-20260907-dsv4pro-ale",
      "model_slug": "deepseek-v4-pro",
      "category": "agentic_computer_use",
      "benchmark_name": "Agents' Last Exam",
      "benchmark_version": null,
      "score_numeric": 25.7,
      "score_display": "25.7%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "DeepSeek Harness",
      "evaluator": "DeepSeek",
      "evaluation_date": "2026-08-13T00:00:00.000Z",
      "source": "https://api-docs.deepseek.com/updates",
      "notes": null
    },
    {
      "id": "deepseek-v4-pro-automationbench-31-8-none-none-none-deepseek",
      "model_slug": "deepseek-v4-pro",
      "category": "agentic_computer_use",
      "benchmark_name": "AutomationBench",
      "benchmark_version": null,
      "score_numeric": 31.8,
      "score_display": "31.8",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "DeepSeek",
      "evaluation_date": null,
      "source": "https://deepseek.com/en/news/v4-preview/",
      "notes": null
    },
    {
      "id": "tq-20260907-dsv4pro-automation",
      "model_slug": "deepseek-v4-pro",
      "category": "agentic_computer_use",
      "benchmark_name": "AutomationBench",
      "benchmark_version": "Public",
      "score_numeric": 31.8,
      "score_display": "31.8%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "DeepSeek Harness",
      "evaluator": "DeepSeek",
      "evaluation_date": "2026-08-13T00:00:00.000Z",
      "source": "https://api-docs.deepseek.com/updates",
      "notes": null
    },
    {
      "id": "tq-20260907-dsv4pro-deepswe",
      "model_slug": "deepseek-v4-pro",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": null,
      "score_numeric": 62.7,
      "score_display": "62.7%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "max",
      "harness": "DeepSeek Harness",
      "evaluator": "DeepSeek",
      "evaluation_date": "2026-08-13T00:00:00.000Z",
      "source": "https://api-docs.deepseek.com/updates",
      "notes": null
    },
    {
      "id": "deepseek-v4-pro-humanity-s-last-exam-42-7-false-none-none-deepseek",
      "model_slug": "deepseek-v4-pro",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 42.7,
      "score_display": "42.7",
      "score_unit": "%",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "DeepSeek",
      "evaluation_date": null,
      "source": "https://deepseek.com/en/news/v4-preview/",
      "notes": null
    },
    {
      "id": "deepseek-v4-pro-humanity-s-last-exam-60-0-true-none-none-deepseek-tools",
      "model_slug": "deepseek-v4-pro",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 60,
      "score_display": "60.0",
      "score_unit": "%",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "DeepSeek",
      "evaluation_date": null,
      "source": "https://deepseek.com/en/news/v4-preview/",
      "notes": null
    },
    {
      "id": "tq-20260907-dsv4pro-hle-no-tools",
      "model_slug": "deepseek-v4-pro",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 42.7,
      "score_display": "42.7%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "max",
      "harness": "DeepSeek V4 Pro direct evaluation",
      "evaluator": "DeepSeek",
      "evaluation_date": "2026-08-13T00:00:00.000Z",
      "source": "https://api-docs.deepseek.com/updates",
      "notes": null
    },
    {
      "id": "tq-20260907-dsv4pro-hle-tools",
      "model_slug": "deepseek-v4-pro",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 60,
      "score_display": "60.0%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "DeepSeek V4 Pro direct evaluation",
      "evaluator": "DeepSeek",
      "evaluation_date": "2026-08-13T00:00:00.000Z",
      "source": "https://api-docs.deepseek.com/updates",
      "notes": null
    },
    {
      "id": "deepseek-v4-pro-terminal-bench-2-1-87-9-none-max-deepseek-harness-deepseek",
      "model_slug": "deepseek-v4-pro",
      "category": "coding",
      "benchmark_name": "Terminal-Bench",
      "benchmark_version": "2.1",
      "score_numeric": 87.9,
      "score_display": "87.9",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": "max",
      "harness": "DeepSeek Harness",
      "evaluator": "DeepSeek",
      "evaluation_date": null,
      "source": "https://deepseek.com/en/news/v4-preview/",
      "notes": null
    },
    {
      "id": "tq-20260907-dsv4pro-terminal21",
      "model_slug": "deepseek-v4-pro",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": null,
      "score_numeric": 87.9,
      "score_display": "87.9%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "max",
      "harness": "DeepSeek Harness",
      "evaluator": "DeepSeek",
      "evaluation_date": "2026-08-13T00:00:00.000Z",
      "source": "https://api-docs.deepseek.com/updates",
      "notes": null
    },
    {
      "id": "deepseek-v4-pro-toolathlon-74-1-none-none-none-deepseek",
      "model_slug": "deepseek-v4-pro",
      "category": "agentic_computer_use",
      "benchmark_name": "Toolathlon",
      "benchmark_version": null,
      "score_numeric": 74.1,
      "score_display": "74.1",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "DeepSeek",
      "evaluation_date": null,
      "source": "https://deepseek.com/en/news/v4-preview/",
      "notes": null
    },
    {
      "id": "tq-20260907-dsv4pro-toolathlon",
      "model_slug": "deepseek-v4-pro",
      "category": "agentic_computer_use",
      "benchmark_name": "Toolathlon",
      "benchmark_version": "Verified",
      "score_numeric": 74.1,
      "score_display": "74.1%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "DeepSeek Harness",
      "evaluator": "DeepSeek",
      "evaluation_date": "2026-08-13T00:00:00.000Z",
      "source": "https://api-docs.deepseek.com/updates",
      "notes": null
    },
    {
      "id": "ernie-5-1-aime-2026-99-6-true-none-none-baidu",
      "model_slug": "ernie-5-1",
      "category": "math_reasoning",
      "benchmark_name": "AIME",
      "benchmark_version": "2026",
      "score_numeric": 99.6,
      "score_display": "99.6",
      "score_unit": "%",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Baidu",
      "evaluation_date": null,
      "source": "https://ernie.baidu.com/blog/posts/ernie-5.1-0508-release/",
      "notes": "Tool-enabled result; not directly comparable with no-tool AIME scores."
    },
    {
      "id": "tq-20260907-036-b1183b81081421319e68",
      "model_slug": "ernie-5-1",
      "category": "math_reasoning",
      "benchmark_name": "AIME",
      "benchmark_version": "2026",
      "score_numeric": 99.6,
      "score_display": "99.6%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Baidu ERNIE 5.1 launch evaluation",
      "evaluator": "Baidu",
      "evaluation_date": "2026-05-09T00:00:00.000Z",
      "source": "https://ernie.baidu.com/blog/posts/ernie-5.1-0508-release/",
      "notes": "Tool-augmented AIME26."
    },
    {
      "id": "tq-20260907-036-a1891b10838ca53df2b1",
      "model_slug": "ernie-5-1",
      "category": "agentic_computer_use",
      "benchmark_name": "Arena Search",
      "benchmark_version": null,
      "score_numeric": 1223,
      "score_display": "1223 Elo",
      "score_unit": "elo",
      "tools": true,
      "reasoning_effort": null,
      "harness": "LMArena Search leaderboard",
      "evaluator": "LMArena",
      "evaluation_date": "2026-05-09T00:00:00.000Z",
      "source": "https://ernie.baidu.com/blog/posts/ernie-5.1-0508-release/",
      "notes": "Baidu reports rank #4 globally and #1 among Chinese models on May 9, 2026."
    },
    {
      "id": "tq-20260906-benchfill-g31lite-gdpv2",
      "model_slug": "gemini-3-1-flash-lite",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA v2",
      "benchmark_version": null,
      "score_numeric": 642,
      "score_display": "642 Elo",
      "score_unit": "Elo",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Artificial Analysis",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-07-21T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-5-flash-lite/",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-g31lite-mle",
      "model_slug": "gemini-3-1-flash-lite",
      "category": "coding",
      "benchmark_name": "MLE-Bench",
      "benchmark_version": null,
      "score_numeric": 22,
      "score_display": "22.0%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-07-21T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-5-flash-lite/",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-g31lite-osworld-verified",
      "model_slug": "gemini-3-1-flash-lite",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld-Verified",
      "benchmark_version": null,
      "score_numeric": 54.3,
      "score_display": "54.3%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-07-21T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-5-flash-lite/",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-g31lite-swe-pro",
      "model_slug": "gemini-3-1-flash-lite",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": "Public",
      "score_numeric": 38.3,
      "score_display": "38.3%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-07-21T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-5-flash-lite/",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-g31lite-terminal21",
      "model_slug": "gemini-3-1-flash-lite",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": "Terminus-2",
      "score_numeric": 31,
      "score_display": "31.0%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Terminus-2",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-07-21T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-5-flash-lite/",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-g31pro-arcagi2",
      "model_slug": "gemini-3-1-pro",
      "category": "math_reasoning",
      "benchmark_name": "ARC-AGI",
      "benchmark_version": "2",
      "score_numeric": 77.1,
      "score_display": "77.1%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "high",
      "harness": "ARC Prize Verified",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-02-19T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-1-pro/",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-g31pro-browsecomp",
      "model_slug": "gemini-3-1-pro",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": null,
      "score_numeric": 85.9,
      "score_display": "85.9%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Search + Python + Browse",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-02-19T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-1-pro/",
      "notes": null
    },
    {
      "id": "gdm-202607-g31pro-deepswe",
      "model_slug": "gemini-3-1-pro",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": null,
      "score_numeric": 12,
      "score_display": "12%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Gemini 3.6 Flash model-card comparison",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-07-21T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-6-flash/",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-g31pro-gdpval-aa",
      "model_slug": "gemini-3-1-pro",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA",
      "benchmark_version": null,
      "score_numeric": 1314,
      "score_display": "1314 Elo",
      "score_unit": "Elo",
      "tools": true,
      "reasoning_effort": "high",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-05-19T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-5-flash/",
      "notes": "Later Google comparison value; distinct from GDPval-AA v2."
    },
    {
      "id": "gdm-202607-g31pro-gdpv2",
      "model_slug": "gemini-3-1-pro",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA v2",
      "benchmark_version": null,
      "score_numeric": 965,
      "score_display": "965 Elo",
      "score_unit": "Elo",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Artificial Analysis",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-07-21T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-6-flash/",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-g31pro-gpqa",
      "model_slug": "gemini-3-1-pro",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 94.3,
      "score_display": "94.3%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "high",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-02-19T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-1-pro/",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-g31pro-hle-no-tools",
      "model_slug": "gemini-3-1-pro",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 44.4,
      "score_display": "44.4%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "high",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-02-19T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-1-pro/",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-g31pro-hle-tools",
      "model_slug": "gemini-3-1-pro",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 51.4,
      "score_display": "51.4%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Search (blocklist) + Code",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-02-19T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-1-pro/",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-g31pro-mcp-atlas",
      "model_slug": "gemini-3-1-pro",
      "category": "agentic_computer_use",
      "benchmark_name": "MCP Atlas",
      "benchmark_version": null,
      "score_numeric": 78.2,
      "score_display": "78.2%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "high",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-05-19T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-5-flash/",
      "notes": null
    },
    {
      "id": "gdm-202607-g31pro-mle",
      "model_slug": "gemini-3-1-pro",
      "category": "coding",
      "benchmark_name": "MLE-Bench",
      "benchmark_version": null,
      "score_numeric": 42.6,
      "score_display": "42.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Gemini 3.6 Flash model-card comparison",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-07-21T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-6-flash/",
      "notes": null
    },
    {
      "id": "gdm-202607-g31pro-osworld-verified",
      "model_slug": "gemini-3-1-pro",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld-Verified",
      "benchmark_version": null,
      "score_numeric": 76.2,
      "score_display": "76.2%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Gemini 3.6 Flash model-card comparison",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-07-21T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-6-flash/",
      "notes": null
    },
    {
      "id": "gdm-202607-g31pro-swe-pro",
      "model_slug": "gemini-3-1-pro",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": "Public",
      "score_numeric": 54.2,
      "score_display": "54.2%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Gemini 3.6 Flash model-card comparison",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-07-21T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-6-flash/",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-g31pro-swe-verified",
      "model_slug": "gemini-3-1-pro",
      "category": "coding",
      "benchmark_name": "SWE-bench Verified",
      "benchmark_version": null,
      "score_numeric": 80.6,
      "score_display": "80.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "high",
      "harness": "Single attempt",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-02-19T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-1-pro/",
      "notes": null
    },
    {
      "id": "gdm-202607-g31pro-terminal21",
      "model_slug": "gemini-3-1-pro",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": "Terminus-2",
      "score_numeric": 73.8,
      "score_display": "73.8%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Terminus-2",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-07-21T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-6-flash/",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-g35-arcagi2",
      "model_slug": "gemini-3-5-flash",
      "category": "math_reasoning",
      "benchmark_name": "ARC-AGI",
      "benchmark_version": "2",
      "score_numeric": 72.1,
      "score_display": "72.1%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-05-19T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-5-flash/",
      "notes": null
    },
    {
      "id": "gdm-202607-g35-deepswe",
      "model_slug": "gemini-3-5-flash",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": null,
      "score_numeric": 37,
      "score_display": "37%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Gemini 3.6 Flash model-card comparison",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-07-21T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-6-flash/",
      "notes": null
    },
    {
      "id": "gemini-3-5-flash-gdpval-aa-1656-none-none-none-google-deepmind",
      "model_slug": "gemini-3-5-flash",
      "category": "professional",
      "benchmark_name": "GDPval-AA",
      "benchmark_version": null,
      "score_numeric": 1656,
      "score_display": "1656",
      "score_unit": "Elo",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": null,
      "source": "https://deepmind.google/models/gemini/",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-g35-gdpval-aa",
      "model_slug": "gemini-3-5-flash",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA",
      "benchmark_version": null,
      "score_numeric": 1656,
      "score_display": "1656 Elo",
      "score_unit": "Elo",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-05-19T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-5-flash/",
      "notes": "Original GDPval-AA benchmark; distinct from GDPval-AA v2."
    },
    {
      "id": "gdm-202607-g35-gdpv2",
      "model_slug": "gemini-3-5-flash",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA v2",
      "benchmark_version": null,
      "score_numeric": 1349,
      "score_display": "1349 Elo",
      "score_unit": "Elo",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Artificial Analysis",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-07-21T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-6-flash/",
      "notes": "Do not conflate with the May launch GDPval-AA result of 1656 Elo; this is GDPVal-AA v2."
    },
    {
      "id": "tq-20260906-benchfill-g35-hle",
      "model_slug": "gemini-3-5-flash",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 40.2,
      "score_display": "40.2%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-05-19T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-5-flash/",
      "notes": null
    },
    {
      "id": "gemini-3-5-flash-mcp-atlas-83-6-none-none-none-google-deepmind",
      "model_slug": "gemini-3-5-flash",
      "category": "agentic_computer_use",
      "benchmark_name": "MCP Atlas",
      "benchmark_version": null,
      "score_numeric": 83.6,
      "score_display": "83.6",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": null,
      "source": "https://deepmind.google/models/gemini/",
      "notes": null
    },
    {
      "id": "google-202605-g35-mcp-atlas",
      "model_slug": "gemini-3-5-flash",
      "category": "agentic_computer_use",
      "benchmark_name": "MCP Atlas",
      "benchmark_version": null,
      "score_numeric": 83.6,
      "score_display": "83.6%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Gemini 3.5 launch evaluation",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-05-19T00:00:00.000Z",
      "source": "https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-5/",
      "notes": null
    },
    {
      "id": "gdm-202607-g35-mle",
      "model_slug": "gemini-3-5-flash",
      "category": "coding",
      "benchmark_name": "MLE-Bench",
      "benchmark_version": null,
      "score_numeric": 49.7,
      "score_display": "49.7%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Gemini 3.6 Flash model-card comparison",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-07-21T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-6-flash/",
      "notes": null
    },
    {
      "id": "gdm-202607-g35-osworld-verified",
      "model_slug": "gemini-3-5-flash",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld-Verified",
      "benchmark_version": null,
      "score_numeric": 78.4,
      "score_display": "78.4%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Gemini 3.6 Flash model-card comparison",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-07-21T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-6-flash/",
      "notes": null
    },
    {
      "id": "gdm-202607-g35-swe-pro",
      "model_slug": "gemini-3-5-flash",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": "Public",
      "score_numeric": 55.1,
      "score_display": "55.1%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Gemini 3.6 Flash model-card comparison",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-07-21T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-6-flash/",
      "notes": null
    },
    {
      "id": "gemini-3-5-flash-terminal-bench-2-1-76-2-none-none-none-google-deepmind",
      "model_slug": "gemini-3-5-flash",
      "category": "coding",
      "benchmark_name": "Terminal-Bench",
      "benchmark_version": "2.1",
      "score_numeric": 76.2,
      "score_display": "76.2",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": null,
      "source": "https://deepmind.google/models/gemini/",
      "notes": null
    },
    {
      "id": "gdm-202605-g35-terminal21",
      "model_slug": "gemini-3-5-flash",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": null,
      "score_numeric": 76.2,
      "score_display": "76.2%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Gemini 3.5 launch evaluation",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-05-19T00:00:00.000Z",
      "source": "https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-5/",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-g35-toolathlon",
      "model_slug": "gemini-3-5-flash",
      "category": "agentic_computer_use",
      "benchmark_name": "Toolathlon",
      "benchmark_version": null,
      "score_numeric": 56.5,
      "score_display": "56.5%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-05-19T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-5-flash/",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-g35lite-gdpv2",
      "model_slug": "gemini-3-5-flash-lite",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA v2",
      "benchmark_version": null,
      "score_numeric": 1140,
      "score_display": "1140 Elo",
      "score_unit": "Elo",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Artificial Analysis",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-07-21T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-5-flash-lite/",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-g35lite-mle",
      "model_slug": "gemini-3-5-flash-lite",
      "category": "coding",
      "benchmark_name": "MLE-Bench",
      "benchmark_version": null,
      "score_numeric": 39.2,
      "score_display": "39.2%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-07-21T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-5-flash-lite/",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-g35lite-osworld-verified",
      "model_slug": "gemini-3-5-flash-lite",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld-Verified",
      "benchmark_version": null,
      "score_numeric": 74,
      "score_display": "74.0%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-07-21T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-5-flash-lite/",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-g35lite-swe-pro",
      "model_slug": "gemini-3-5-flash-lite",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": "Public",
      "score_numeric": 54.2,
      "score_display": "54.2%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-07-21T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-5-flash-lite/",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-g35lite-terminal21",
      "model_slug": "gemini-3-5-flash-lite",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": "Terminus-2",
      "score_numeric": 54,
      "score_display": "54.0%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Terminus-2",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-07-21T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-5-flash-lite/",
      "notes": null
    },
    {
      "id": "gdm-202608-g36-agents-last-exam",
      "model_slug": "gemini-3-6-flash",
      "category": "agentic_computer_use",
      "benchmark_name": "Agents' Last Exam",
      "benchmark_version": null,
      "score_numeric": 24.2,
      "score_display": "24.2%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Gemini 3.7 Flash model-card comparison",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-08-13T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
      "notes": "Google's table labels this benchmark Agent's Last Exam; normalized here to the catalog label."
    },
    {
      "id": "gdm-202608-g36-automation",
      "model_slug": "gemini-3-6-flash",
      "category": "agentic_computer_use",
      "benchmark_name": "AutomationBench",
      "benchmark_version": "Private set",
      "score_numeric": 17,
      "score_display": "17.0%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Gemini 3.7 Flash model-card comparison",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-08-13T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
      "notes": null
    },
    {
      "id": "gdm-202608-g36-deepswe",
      "model_slug": "gemini-3-6-flash",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": null,
      "score_numeric": 48.6,
      "score_display": "48.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Gemini 3.7 Flash model-card comparison",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-08-13T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
      "notes": null
    },
    {
      "id": "gdm-202608-g36-frontiercode-main",
      "model_slug": "gemini-3-6-flash",
      "category": "coding",
      "benchmark_name": "FrontierCode 1.1 Main",
      "benchmark_version": null,
      "score_numeric": 34.4,
      "score_display": "34.4%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Gemini 3.7 Flash model-card comparison",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-08-13T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
      "notes": null
    },
    {
      "id": "gdm-202608-g36-gdpv2",
      "model_slug": "gemini-3-6-flash",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA v2",
      "benchmark_version": null,
      "score_numeric": 1422,
      "score_display": "1422 Elo",
      "score_unit": "Elo",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Artificial Analysis",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-08-13T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
      "notes": null
    },
    {
      "id": "gdm-202608-g36-hleverified",
      "model_slug": "gemini-3-6-flash",
      "category": "knowledge",
      "benchmark_name": "HLE-Verified",
      "benchmark_version": null,
      "score_numeric": 51.2,
      "score_display": "51.2%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Gemini 3.7 Flash model-card comparison",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-08-13T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
      "notes": null
    },
    {
      "id": "gdm-202607-g36-mle",
      "model_slug": "gemini-3-6-flash",
      "category": "coding",
      "benchmark_name": "MLE-Bench",
      "benchmark_version": null,
      "score_numeric": 63.9,
      "score_display": "63.9%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Gemini 3.6 Flash model card",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-07-21T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-6-flash/",
      "notes": null
    },
    {
      "id": "gdm-202608-g36-osworld20",
      "model_slug": "gemini-3-6-flash",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld 2.0",
      "benchmark_version": null,
      "score_numeric": 33.8,
      "score_display": "33.8%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Gemini 3.7 Flash model-card comparison",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-08-13T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
      "notes": null
    },
    {
      "id": "gdm-202607-g36-osworld-verified",
      "model_slug": "gemini-3-6-flash",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld-Verified",
      "benchmark_version": null,
      "score_numeric": 83,
      "score_display": "83.0%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Gemini 3.6 Flash model card",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-07-21T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-6-flash/",
      "notes": null
    },
    {
      "id": "gdm-202607-g36-swe-pro",
      "model_slug": "gemini-3-6-flash",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": "Public",
      "score_numeric": 58.7,
      "score_display": "58.7%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Gemini 3.6 Flash model card",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-07-21T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-6-flash/",
      "notes": null
    },
    {
      "id": "gdm-202608-g36-terminal21",
      "model_slug": "gemini-3-6-flash",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": null,
      "score_numeric": 78,
      "score_display": "78.0%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Gemini 3.7 Flash model-card comparison",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-08-13T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
      "notes": null
    },
    {
      "id": "gdm-202608-g36-terminal30",
      "model_slug": "gemini-3-6-flash",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 3.0",
      "benchmark_version": null,
      "score_numeric": 5.4,
      "score_display": "5.4%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Gemini 3.7 Flash model-card comparison",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-08-13T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
      "notes": null
    },
    {
      "id": "gdm-202608-g37-agents-last-exam",
      "model_slug": "gemini-3-7-flash",
      "category": "agentic_computer_use",
      "benchmark_name": "Agents' Last Exam",
      "benchmark_version": null,
      "score_numeric": 26.3,
      "score_display": "26.3%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Gemini 3.7 Flash model card",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-08-13T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
      "notes": "Google's table labels this benchmark Agent's Last Exam; normalized here to the catalog label."
    },
    {
      "id": "gdm-202608-g37-automation",
      "model_slug": "gemini-3-7-flash",
      "category": "agentic_computer_use",
      "benchmark_name": "AutomationBench",
      "benchmark_version": "Private set",
      "score_numeric": 30.4,
      "score_display": "30.4%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Gemini 3.7 Flash model card",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-08-13T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-g37-cursorbench32",
      "model_slug": "gemini-3-7-flash",
      "category": "coding",
      "benchmark_name": "CursorBench",
      "benchmark_version": "3.2",
      "score_numeric": 61.6,
      "score_display": "61.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "high",
      "harness": "CursorBench 3.2",
      "evaluator": "Cursor",
      "evaluation_date": "2026-09-02T00:00:00.000Z",
      "source": "https://cursor.com/evals",
      "notes": null
    },
    {
      "id": "gemini-3-7-flash-deepswe-1-1-65-3-none-none-none-google-deepmind",
      "model_slug": "gemini-3-7-flash",
      "category": "coding",
      "benchmark_name": "DeepSWE",
      "benchmark_version": "1.1",
      "score_numeric": 65.3,
      "score_display": "65.3",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": null,
      "source": "https://blog.google/innovation-and-ai/models-and-research/gemini-models/introducing-gemini-3-7-flash/",
      "notes": null
    },
    {
      "id": "gdm-202608-g37-deepswe",
      "model_slug": "gemini-3-7-flash",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": null,
      "score_numeric": 65.3,
      "score_display": "65.3%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Gemini 3.7 Flash model card",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-08-13T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
      "notes": null
    },
    {
      "id": "gemini-3-7-flash-frontiercode-1-1-43-6-none-none-none-google-deepmind",
      "model_slug": "gemini-3-7-flash",
      "category": "coding",
      "benchmark_name": "FrontierCode",
      "benchmark_version": "1.1",
      "score_numeric": 43.6,
      "score_display": "43.6",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": null,
      "source": "https://blog.google/innovation-and-ai/models-and-research/gemini-models/introducing-gemini-3-7-flash/",
      "notes": null
    },
    {
      "id": "gdm-202608-g37-frontiercode-main",
      "model_slug": "gemini-3-7-flash",
      "category": "coding",
      "benchmark_name": "FrontierCode 1.1 Main",
      "benchmark_version": null,
      "score_numeric": 43.6,
      "score_display": "43.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Gemini 3.7 Flash model card",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-08-13T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
      "notes": null
    },
    {
      "id": "gdm-202608-g37-gdpv2",
      "model_slug": "gemini-3-7-flash",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA v2",
      "benchmark_version": null,
      "score_numeric": 1525,
      "score_display": "1525 Elo",
      "score_unit": "Elo",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Artificial Analysis",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-08-13T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
      "notes": null
    },
    {
      "id": "gdm-202608-g37-hleverified",
      "model_slug": "gemini-3-7-flash",
      "category": "knowledge",
      "benchmark_name": "HLE-Verified",
      "benchmark_version": null,
      "score_numeric": 53.6,
      "score_display": "53.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Gemini 3.7 Flash model card",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-08-13T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-g37-livecodebench",
      "model_slug": "gemini-3-7-flash",
      "category": "coding",
      "benchmark_name": "LiveCodeBench",
      "benchmark_version": null,
      "score_numeric": 88.7,
      "score_display": "88.7%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "high",
      "harness": null,
      "evaluator": "Vals AI",
      "evaluation_date": "2026-09-05T00:00:00.000Z",
      "source": "https://www.vals.ai/benchmarks/lcb",
      "notes": null
    },
    {
      "id": "gdm-202608-g37-osworld20",
      "model_slug": "gemini-3-7-flash",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld 2.0",
      "benchmark_version": null,
      "score_numeric": 47.9,
      "score_display": "47.9%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Gemini 3.7 Flash model card",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-08-13T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
      "notes": null
    },
    {
      "id": "gdm-202608-g37-terminal21",
      "model_slug": "gemini-3-7-flash",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": null,
      "score_numeric": 85.8,
      "score_display": "85.8%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Gemini 3.7 Flash model card",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-08-13T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
      "notes": null
    },
    {
      "id": "gdm-202608-g37-terminal30",
      "model_slug": "gemini-3-7-flash",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 3.0",
      "benchmark_version": null,
      "score_numeric": 14.9,
      "score_display": "14.9%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Gemini 3.7 Flash model card",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-08-13T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-g37-terminal40",
      "model_slug": "gemini-3-7-flash",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 4.0",
      "benchmark_version": null,
      "score_numeric": 11.2,
      "score_display": "11.2%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-09-02T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-8-flash/",
      "notes": "Reported in the Gemini 3.8 Flash comparison table."
    },
    {
      "id": "tq-20260906-benchfill-g38-cursorbench32",
      "model_slug": "gemini-3-8-flash",
      "category": "coding",
      "benchmark_name": "CursorBench",
      "benchmark_version": "3.2",
      "score_numeric": 69.2,
      "score_display": "69.2%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "high",
      "harness": "CursorBench 3.2",
      "evaluator": "Cursor",
      "evaluation_date": "2026-09-02T00:00:00.000Z",
      "source": "https://cursor.com/evals",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-g38-deepswe",
      "model_slug": "gemini-3-8-flash",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": null,
      "score_numeric": 73.7,
      "score_display": "73.7%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-09-02T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-8-flash/",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-g38-gdpv2",
      "model_slug": "gemini-3-8-flash",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA v2",
      "benchmark_version": null,
      "score_numeric": 1545,
      "score_display": "1545 Elo",
      "score_unit": "Elo",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-09-02T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-8-flash/",
      "notes": null
    },
    {
      "id": "gdm-202609-g38-hleverified",
      "model_slug": "gemini-3-8-flash",
      "category": "knowledge",
      "benchmark_name": "HLE-Verified",
      "benchmark_version": null,
      "score_numeric": 54.9,
      "score_display": "54.9%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Gemini 3.8 Flash launch evaluation",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-09-02T00:00:00.000Z",
      "source": "https://deepmind.google/models/gemini/flash/",
      "notes": null
    },
    {
      "id": "gemini-3-8-flash-hle-verified-54-9-none-none-none-google-deepmind",
      "model_slug": "gemini-3-8-flash",
      "category": "knowledge",
      "benchmark_name": "HLE-Verified",
      "benchmark_version": null,
      "score_numeric": 54.9,
      "score_display": "54.9",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": null,
      "source": "https://ai.google.dev/gemini-api/docs/models/gemini-3.8-flash",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-g38-livecodebench",
      "model_slug": "gemini-3-8-flash",
      "category": "coding",
      "benchmark_name": "LiveCodeBench",
      "benchmark_version": null,
      "score_numeric": 89.5,
      "score_display": "89.5%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "high",
      "harness": null,
      "evaluator": "Vals AI",
      "evaluation_date": "2026-09-05T00:00:00.000Z",
      "source": "https://www.vals.ai/benchmarks/lcb",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-g38-osworld20",
      "model_slug": "gemini-3-8-flash",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld 2.0",
      "benchmark_version": null,
      "score_numeric": 59,
      "score_display": "59.0%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-09-02T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-8-flash/",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-g38-terminal21",
      "model_slug": "gemini-3-8-flash",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": null,
      "score_numeric": 89.4,
      "score_display": "89.4%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-09-02T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-8-flash/",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-g38-terminal40",
      "model_slug": "gemini-3-8-flash",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 4.0",
      "benchmark_version": null,
      "score_numeric": 19.1,
      "score_display": "19.1%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-09-02T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-8-flash/",
      "notes": null
    },
    {
      "id": "tq-20260907-033-681b22a266e5ada2fd0a",
      "model_slug": "gemma-4-12b",
      "category": "math_reasoning",
      "benchmark_name": "AIME",
      "benchmark_version": "2026",
      "score_numeric": 77.5,
      "score_display": "77.5%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-06-03T00:00:00.000Z",
      "source": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "notes": null
    },
    {
      "id": "tq-20260907-033-d23eeba2d54e8a440277",
      "model_slug": "gemma-4-12b",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 78.8,
      "score_display": "78.8%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-06-03T00:00:00.000Z",
      "source": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "notes": null
    },
    {
      "id": "tq-20260907-033-68a9732b94aa640c5dd4",
      "model_slug": "gemma-4-12b",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 5.2,
      "score_display": "5.2%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-06-03T00:00:00.000Z",
      "source": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "notes": null
    },
    {
      "id": "tq-20260907-033-3086e83f65d269e63949",
      "model_slug": "gemma-4-12b",
      "category": "coding",
      "benchmark_name": "LiveCodeBench",
      "benchmark_version": "v6",
      "score_numeric": 72,
      "score_display": "72.0%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-06-03T00:00:00.000Z",
      "source": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "notes": null
    },
    {
      "id": "tq-20260907-033-8c508b7d7864c8ef2c9a",
      "model_slug": "gemma-4-12b",
      "category": "knowledge",
      "benchmark_name": "MMLU-Pro",
      "benchmark_version": null,
      "score_numeric": 77.2,
      "score_display": "77.2%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-06-03T00:00:00.000Z",
      "source": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "notes": null
    },
    {
      "id": "tq-20260907-033-a5c92ec5679238bd3118",
      "model_slug": "gemma-4-26b-a4b",
      "category": "math_reasoning",
      "benchmark_name": "AIME",
      "benchmark_version": "2026",
      "score_numeric": 88.3,
      "score_display": "88.3%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-03-31T00:00:00.000Z",
      "source": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "notes": null
    },
    {
      "id": "tq-20260907-033-6722bfa03e02808e5ec2",
      "model_slug": "gemma-4-26b-a4b",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 82.3,
      "score_display": "82.3%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-03-31T00:00:00.000Z",
      "source": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "notes": null
    },
    {
      "id": "tq-20260907-033-6e1ce7fac3ae6cdae196",
      "model_slug": "gemma-4-26b-a4b",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": "search",
      "score_numeric": 17.2,
      "score_display": "17.2%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-03-31T00:00:00.000Z",
      "source": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "notes": "Google model card reports HLE with search."
    },
    {
      "id": "tq-20260907-033-fd6501d0a13f41af1be8",
      "model_slug": "gemma-4-26b-a4b",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 8.7,
      "score_display": "8.7%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-03-31T00:00:00.000Z",
      "source": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "notes": null
    },
    {
      "id": "tq-20260907-033-04e42df4e687c8a9f6aa",
      "model_slug": "gemma-4-26b-a4b",
      "category": "coding",
      "benchmark_name": "LiveCodeBench",
      "benchmark_version": "v6",
      "score_numeric": 77.1,
      "score_display": "77.1%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-03-31T00:00:00.000Z",
      "source": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "notes": null
    },
    {
      "id": "tq-20260907-033-9a2cd30b0f86aa3eaa06",
      "model_slug": "gemma-4-26b-a4b",
      "category": "knowledge",
      "benchmark_name": "MMLU-Pro",
      "benchmark_version": null,
      "score_numeric": 82.6,
      "score_display": "82.6%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-03-31T00:00:00.000Z",
      "source": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "notes": null
    },
    {
      "id": "gemma-4-31b-aime-2026-89-2-none-it-thinking-none-google-deepmind",
      "model_slug": "gemma-4-31b",
      "category": "math_reasoning",
      "benchmark_name": "AIME",
      "benchmark_version": "2026",
      "score_numeric": 89.2,
      "score_display": "89.2",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": "IT Thinking",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": null,
      "source": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "notes": null
    },
    {
      "id": "tq-20260907-033-44e8a99ee024dbefd580",
      "model_slug": "gemma-4-31b",
      "category": "math_reasoning",
      "benchmark_name": "AIME",
      "benchmark_version": "2026",
      "score_numeric": 89.2,
      "score_display": "89.2%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-03-31T00:00:00.000Z",
      "source": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "notes": null
    },
    {
      "id": "gemma-4-31b-gpqa-diamond-84-3-none-it-thinking-none-google-deepmind",
      "model_slug": "gemma-4-31b",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 84.3,
      "score_display": "84.3",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": "IT Thinking",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": null,
      "source": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "notes": null
    },
    {
      "id": "tq-20260907-033-dbff0b8e942d7347d6ec",
      "model_slug": "gemma-4-31b",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 84.3,
      "score_display": "84.3%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-03-31T00:00:00.000Z",
      "source": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "notes": null
    },
    {
      "id": "tq-20260907-033-75a2747d47eeab36bd5a",
      "model_slug": "gemma-4-31b",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": "search",
      "score_numeric": 26.5,
      "score_display": "26.5%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-03-31T00:00:00.000Z",
      "source": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "notes": "Google model card reports HLE with search."
    },
    {
      "id": "tq-20260907-033-8c8c494dd87882176955",
      "model_slug": "gemma-4-31b",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 19.5,
      "score_display": "19.5%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-03-31T00:00:00.000Z",
      "source": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "notes": null
    },
    {
      "id": "gemma-4-31b-livecodebench-v6-80-0-none-it-thinking-none-google-deepmind",
      "model_slug": "gemma-4-31b",
      "category": "coding",
      "benchmark_name": "LiveCodeBench",
      "benchmark_version": "v6",
      "score_numeric": 80,
      "score_display": "80.0",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": "IT Thinking",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": null,
      "source": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "notes": null
    },
    {
      "id": "tq-20260907-033-7da0455f38b54cc9a400",
      "model_slug": "gemma-4-31b",
      "category": "coding",
      "benchmark_name": "LiveCodeBench",
      "benchmark_version": "v6",
      "score_numeric": 80,
      "score_display": "80.0%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-03-31T00:00:00.000Z",
      "source": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "notes": null
    },
    {
      "id": "tq-20260907-033-7f2951f618b979fc9091",
      "model_slug": "gemma-4-31b",
      "category": "knowledge",
      "benchmark_name": "MMLU-Pro",
      "benchmark_version": null,
      "score_numeric": 85.2,
      "score_display": "85.2%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-03-31T00:00:00.000Z",
      "source": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "notes": null
    },
    {
      "id": "gemma-4-31b-mmmlu-85-2-none-it-thinking-none-google-deepmind",
      "model_slug": "gemma-4-31b",
      "category": "knowledge",
      "benchmark_name": "MMMLU",
      "benchmark_version": null,
      "score_numeric": 85.2,
      "score_display": "85.2",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": "IT Thinking",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": null,
      "source": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "notes": null
    },
    {
      "id": "gemma-4-31b-mmmu-pro-76-9-none-it-thinking-none-google-deepmind",
      "model_slug": "gemma-4-31b",
      "category": "multimodal",
      "benchmark_name": "MMMU Pro",
      "benchmark_version": null,
      "score_numeric": 76.9,
      "score_display": "76.9",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": "IT Thinking",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": null,
      "source": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "notes": null
    },
    {
      "id": "tq-20260907-033-9f582b455c677eae4d4d",
      "model_slug": "gemma-4-31b",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": "Public",
      "score_numeric": 35.7,
      "score_display": "35.7%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Qwen agent scaffold",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-04-21T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.6-27b",
      "notes": "Cross-vendor rerun published by Qwen; not a Google evaluation."
    },
    {
      "id": "tq-20260907-033-2915cef8870bf8970f66",
      "model_slug": "gemma-4-31b",
      "category": "coding",
      "benchmark_name": "SWE-bench Verified",
      "benchmark_version": null,
      "score_numeric": 52,
      "score_display": "52.0%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Qwen agent scaffold",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-04-21T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.6-27b",
      "notes": "Cross-vendor rerun published by Qwen; not a Google evaluation."
    },
    {
      "id": "tq-20260907-033-3b9daa979b8b9f92c931",
      "model_slug": "gemma-4-31b",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.0",
      "benchmark_version": null,
      "score_numeric": 42.9,
      "score_display": "42.9%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Harbor / Terminus-2",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-04-21T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.6-27b",
      "notes": "Cross-vendor rerun published by Qwen; not a Google evaluation."
    },
    {
      "id": "gemma-4-31b-τ²-bench-retail-86-4-none-it-thinking-none-google-deepmind",
      "model_slug": "gemma-4-31b",
      "category": "agentic_computer_use",
      "benchmark_name": "τ²-bench Retail",
      "benchmark_version": null,
      "score_numeric": 86.4,
      "score_display": "86.4",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": "IT Thinking",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": null,
      "source": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "notes": null
    },
    {
      "id": "tq-20260907-033-dc79956b7cf2f1fed62f",
      "model_slug": "gemma-4-e2b",
      "category": "math_reasoning",
      "benchmark_name": "AIME",
      "benchmark_version": "2026",
      "score_numeric": 37.5,
      "score_display": "37.5%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-03-31T00:00:00.000Z",
      "source": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "notes": null
    },
    {
      "id": "tq-20260907-033-432ed4ababad31f52212",
      "model_slug": "gemma-4-e2b",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 43.4,
      "score_display": "43.4%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-03-31T00:00:00.000Z",
      "source": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "notes": null
    },
    {
      "id": "tq-20260907-033-0eac4cd395062ea011b9",
      "model_slug": "gemma-4-e2b",
      "category": "coding",
      "benchmark_name": "LiveCodeBench",
      "benchmark_version": "v6",
      "score_numeric": 44,
      "score_display": "44.0%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-03-31T00:00:00.000Z",
      "source": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "notes": null
    },
    {
      "id": "tq-20260907-033-83b815f715d25103ee4a",
      "model_slug": "gemma-4-e2b",
      "category": "knowledge",
      "benchmark_name": "MMLU-Pro",
      "benchmark_version": null,
      "score_numeric": 60,
      "score_display": "60.0%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-03-31T00:00:00.000Z",
      "source": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "notes": null
    },
    {
      "id": "tq-20260907-033-3e661668bc7182c7f201",
      "model_slug": "gemma-4-e4b",
      "category": "math_reasoning",
      "benchmark_name": "AIME",
      "benchmark_version": "2026",
      "score_numeric": 42.5,
      "score_display": "42.5%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-03-31T00:00:00.000Z",
      "source": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "notes": null
    },
    {
      "id": "tq-20260907-033-306166a5f4637eafbc04",
      "model_slug": "gemma-4-e4b",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 58.6,
      "score_display": "58.6%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-03-31T00:00:00.000Z",
      "source": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "notes": null
    },
    {
      "id": "tq-20260907-033-5ffdce327c859213e632",
      "model_slug": "gemma-4-e4b",
      "category": "coding",
      "benchmark_name": "LiveCodeBench",
      "benchmark_version": "v6",
      "score_numeric": 52,
      "score_display": "52.0%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-03-31T00:00:00.000Z",
      "source": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "notes": null
    },
    {
      "id": "tq-20260907-033-34bfe69737117b6696f5",
      "model_slug": "gemma-4-e4b",
      "category": "knowledge",
      "benchmark_name": "MMLU-Pro",
      "benchmark_version": null,
      "score_numeric": 69.4,
      "score_display": "69.4%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-03-31T00:00:00.000Z",
      "source": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm47f-aime25",
      "model_slug": "glm-4-7-flash",
      "category": "math_reasoning",
      "benchmark_name": "AIME",
      "benchmark_version": "2025",
      "score_numeric": 91.6,
      "score_display": "91.6%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-01-19T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-4.7-Flash",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm47f-browse",
      "model_slug": "glm-4-7-flash",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": null,
      "score_numeric": 42.8,
      "score_display": "42.8%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-01-19T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-4.7-Flash",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm47f-gpqa",
      "model_slug": "glm-4-7-flash",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 75.2,
      "score_display": "75.2%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-01-19T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-4.7-Flash",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm47f-hle",
      "model_slug": "glm-4-7-flash",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 14.4,
      "score_display": "14.4%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-01-19T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-4.7-Flash",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm47f-lcb",
      "model_slug": "glm-4-7-flash",
      "category": "coding",
      "benchmark_name": "LiveCodeBench",
      "benchmark_version": "v6",
      "score_numeric": 64,
      "score_display": "64.0%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-01-19T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-4.7-Flash",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm47f-swev",
      "model_slug": "glm-4-7-flash",
      "category": "coding",
      "benchmark_name": "SWE-bench Verified",
      "benchmark_version": null,
      "score_numeric": 59.2,
      "score_display": "59.2%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-01-19T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-4.7-Flash",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm47f-tau2",
      "model_slug": "glm-4-7-flash",
      "category": "agentic_computer_use",
      "benchmark_name": "τ²-Bench",
      "benchmark_version": null,
      "score_numeric": 79.5,
      "score_display": "79.5%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-01-19T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-4.7-Flash",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm5-aime26",
      "model_slug": "glm-5",
      "category": "math_reasoning",
      "benchmark_name": "AIME",
      "benchmark_version": "2026 I",
      "score_numeric": 92.7,
      "score_display": "92.7%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-02-12T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-5",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm5-browse",
      "model_slug": "glm-5",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": null,
      "score_numeric": 62,
      "score_display": "62.0%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-02-12T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-5",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm5-browse-cm",
      "model_slug": "glm-5",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": "Context management",
      "score_numeric": 75.9,
      "score_display": "75.9%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-02-12T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-5",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm5-gpqa",
      "model_slug": "glm-5",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 86,
      "score_display": "86.0%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-02-12T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-5",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm5-hmmt",
      "model_slug": "glm-5",
      "category": "math_reasoning",
      "benchmark_name": "HMMT",
      "benchmark_version": "Nov. 2025",
      "score_numeric": 96.9,
      "score_display": "96.9%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-02-12T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-5",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm5-hle-no",
      "model_slug": "glm-5",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 30.5,
      "score_display": "30.5%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-02-12T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-5",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm5-hle-tools",
      "model_slug": "glm-5",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 50.4,
      "score_display": "50.4%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-02-12T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-5",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm5-mcp",
      "model_slug": "glm-5",
      "category": "agentic_computer_use",
      "benchmark_name": "MCP Atlas",
      "benchmark_version": "Public Set",
      "score_numeric": 67.8,
      "score_display": "67.8%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-02-12T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-5",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm5-swev",
      "model_slug": "glm-5",
      "category": "coding",
      "benchmark_name": "SWE-bench Verified",
      "benchmark_version": null,
      "score_numeric": 77.8,
      "score_display": "77.8%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenHands",
      "evaluator": "Z.ai",
      "evaluation_date": "2026-02-12T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-5",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm5-term20",
      "model_slug": "glm-5",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.0",
      "benchmark_version": "Verified",
      "score_numeric": 61.1,
      "score_display": "61.1%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Claude Code",
      "evaluator": "Z.ai",
      "evaluation_date": "2026-02-12T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-5",
      "notes": "Uses the verified Terminal-Bench variant; higher of the two official GLM-5 harness rows."
    },
    {
      "id": "tq-20260907-031-glm5-tau2",
      "model_slug": "glm-5",
      "category": "agentic_computer_use",
      "benchmark_name": "τ²-Bench",
      "benchmark_version": null,
      "score_numeric": 89.7,
      "score_display": "89.7%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-02-12T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-5",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm51-aime",
      "model_slug": "glm-5-1",
      "category": "math_reasoning",
      "benchmark_name": "AIME",
      "benchmark_version": "2026",
      "score_numeric": 95.3,
      "score_display": "95.3%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-04-07T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-5.1",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm51-browse",
      "model_slug": "glm-5-1",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": null,
      "score_numeric": 68,
      "score_display": "68.0%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-04-07T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-5.1",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm51-browse-cm",
      "model_slug": "glm-5-1",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": "Context management",
      "score_numeric": 79.3,
      "score_display": "79.3%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-04-07T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-5.1",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm51-gpqa",
      "model_slug": "glm-5-1",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 86.2,
      "score_display": "86.2%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-04-07T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-5.1",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm51-hmmt-feb",
      "model_slug": "glm-5-1",
      "category": "math_reasoning",
      "benchmark_name": "HMMT",
      "benchmark_version": "Feb. 2026",
      "score_numeric": 82.6,
      "score_display": "82.6%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-04-07T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-5.1",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm51-hmmt-nov",
      "model_slug": "glm-5-1",
      "category": "math_reasoning",
      "benchmark_name": "HMMT",
      "benchmark_version": "Nov. 2025",
      "score_numeric": 94,
      "score_display": "94.0%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-04-07T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-5.1",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm51-hle-no",
      "model_slug": "glm-5-1",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 31,
      "score_display": "31.0%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-04-07T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-5.1",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm51-hle-tools",
      "model_slug": "glm-5-1",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 52.3,
      "score_display": "52.3%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-04-07T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-5.1",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm51-mcp",
      "model_slug": "glm-5-1",
      "category": "agentic_computer_use",
      "benchmark_name": "MCP Atlas",
      "benchmark_version": "Public Set",
      "score_numeric": 71.8,
      "score_display": "71.8%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-04-07T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-5.1",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm51-swepro",
      "model_slug": "glm-5-1",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": "Public",
      "score_numeric": 58.4,
      "score_display": "58.4%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-04-07T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-5.1",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm51-term20",
      "model_slug": "glm-5-1",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.0",
      "benchmark_version": null,
      "score_numeric": 63.5,
      "score_display": "63.5%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Terminus-2",
      "evaluator": "Z.ai",
      "evaluation_date": "2026-04-07T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-5.1",
      "notes": null
    },
    {
      "id": "tq-20260907-glm52-agents-last-exam",
      "model_slug": "glm-5-2",
      "category": "agentic_computer_use",
      "benchmark_name": "Agents' Last Exam",
      "benchmark_version": null,
      "score_numeric": 20.4,
      "score_display": "20.4%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-09-06T00:00:00.000Z",
      "source": "https://autoclaw.z.ai/blog/model/glm-5.3-flash/",
      "notes": null
    },
    {
      "id": "tq-20260907-glm52-automation",
      "model_slug": "glm-5-2",
      "category": "agentic_computer_use",
      "benchmark_name": "AutomationBench",
      "benchmark_version": "v1.0.6",
      "score_numeric": 26.2,
      "score_display": "26.2%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-09-06T00:00:00.000Z",
      "source": "https://autoclaw.z.ai/blog/model/glm-5.3-flash/",
      "notes": null
    },
    {
      "id": "tq-20260907-glm52-deepswe",
      "model_slug": "glm-5-2",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": null,
      "score_numeric": 46.2,
      "score_display": "46.2%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-09-06T00:00:00.000Z",
      "source": "https://autoclaw.z.ai/blog/model/glm-5.3-flash/",
      "notes": "GLM-5.2 comparison value in Z.ai's GLM-5.3-Flash evaluation table."
    },
    {
      "id": "tq-20260907-glm52-gdpv2",
      "model_slug": "glm-5-2",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA v2",
      "benchmark_version": null,
      "score_numeric": 1504,
      "score_display": "1504 Elo",
      "score_unit": "Elo",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Artificial Analysis",
      "evaluator": "Z.ai",
      "evaluation_date": "2026-09-06T00:00:00.000Z",
      "source": "https://autoclaw.z.ai/blog/model/glm-5.3-flash/",
      "notes": null
    },
    {
      "id": "tq-20260907-glm52-hle-tools",
      "model_slug": "glm-5-2",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 54.7,
      "score_display": "54.7%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-09-06T00:00:00.000Z",
      "source": "https://autoclaw.z.ai/blog/model/glm-5.3-flash/",
      "notes": null
    },
    {
      "id": "tq-20260907-glm52-swepro",
      "model_slug": "glm-5-2",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": "Public",
      "score_numeric": 62.1,
      "score_display": "62.1%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-06-24T00:00:00.000Z",
      "source": "https://docs.z.ai/guides/llm/glm-5.2",
      "notes": null
    },
    {
      "id": "tq-20260907-glm52-terminal21",
      "model_slug": "glm-5-2",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": null,
      "score_numeric": 81,
      "score_display": "81.0%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Claude Code",
      "evaluator": "Z.ai",
      "evaluation_date": "2026-06-24T00:00:00.000Z",
      "source": "https://docs.z.ai/guides/llm/glm-5.2",
      "notes": null
    },
    {
      "id": "tq-20260907-glm52-toolathlon",
      "model_slug": "glm-5-2",
      "category": "agentic_computer_use",
      "benchmark_name": "Toolathlon",
      "benchmark_version": "Verified",
      "score_numeric": 59.9,
      "score_display": "59.9%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-09-06T00:00:00.000Z",
      "source": "https://autoclaw.z.ai/blog/model/glm-5.3-flash/",
      "notes": null
    },
    {
      "id": "tq-20260907-glm53-ale",
      "model_slug": "glm-5-3",
      "category": "agentic_computer_use",
      "benchmark_name": "Agents' Last Exam",
      "benchmark_version": "ALE-CLI",
      "score_numeric": 28.5,
      "score_display": "28.5%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "Claude Code; 1M context; 64K output",
      "evaluator": "Z.ai",
      "evaluation_date": "2026-08-28T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-5.3",
      "notes": null
    },
    {
      "id": "tq-20260907-glm53-automation",
      "model_slug": "glm-5-3",
      "category": "agentic_computer_use",
      "benchmark_name": "AutomationBench",
      "benchmark_version": "v1.0.6",
      "score_numeric": 48.2,
      "score_display": "48.2%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "GLM-5.3 official reproduction",
      "evaluator": "Z.ai",
      "evaluation_date": "2026-08-28T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-5.3",
      "notes": null
    },
    {
      "id": "tq-20260907-glm53-deepswe",
      "model_slug": "glm-5-3",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": null,
      "score_numeric": 66.9,
      "score_display": "66.9%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "max",
      "harness": "mini-swe-agent; 400K context",
      "evaluator": "Z.ai",
      "evaluation_date": "2026-08-28T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-5.3",
      "notes": null
    },
    {
      "id": "tq-20260907-glm53-gdpv2",
      "model_slug": "glm-5-3",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA v2",
      "benchmark_version": null,
      "score_numeric": 1769,
      "score_display": "1769 Elo",
      "score_unit": "Elo",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "Artificial Analysis",
      "evaluator": "Artificial Analysis",
      "evaluation_date": "2026-08-28T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-5.3",
      "notes": "Reported in Z.ai's official model-card comparison table."
    },
    {
      "id": "tq-20260907-glm53-hle-tools",
      "model_slug": "glm-5-3",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 62.5,
      "score_display": "62.5%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "300K context with context management",
      "evaluator": "Z.ai",
      "evaluation_date": "2026-08-28T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-5.3",
      "notes": "Official footnote uses GPT-5.6 Luna medium as judge."
    },
    {
      "id": "tq-20260907-glm53-terminal21",
      "model_slug": "glm-5-3",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": null,
      "score_numeric": 88.2,
      "score_display": "88.2%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "max",
      "harness": "GLM-5.3 official reproduction",
      "evaluator": "Z.ai",
      "evaluation_date": "2026-08-28T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-5.3",
      "notes": null
    },
    {
      "id": "tq-20260907-glm53-terminal30",
      "model_slug": "glm-5-3",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 3.0",
      "benchmark_version": null,
      "score_numeric": 28.3,
      "score_display": "28.3%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "max",
      "harness": "GLM-5.3 official reproduction",
      "evaluator": "Z.ai",
      "evaluation_date": "2026-08-28T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-5.3",
      "notes": null
    },
    {
      "id": "tq-20260907-glm53-toolathlon",
      "model_slug": "glm-5-3",
      "category": "agentic_computer_use",
      "benchmark_name": "Toolathlon",
      "benchmark_version": "Verified",
      "score_numeric": 73,
      "score_display": "73.0%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "Official evaluation service; pass@1 average of 3 runs",
      "evaluator": "Z.ai",
      "evaluation_date": "2026-08-28T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-5.3",
      "notes": null
    },
    {
      "id": "glm-5-3-flash-agents-last-exam-26-3-none-none-none-z-ai",
      "model_slug": "glm-5-3-flash",
      "category": "agentic_computer_use",
      "benchmark_name": "Agents' Last Exam",
      "benchmark_version": null,
      "score_numeric": 26.3,
      "score_display": "26.3",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": null,
      "source": "https://autoclaw.z.ai/blog/model/glm-5.3-flash/",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm53f-ale",
      "model_slug": "glm-5-3-flash",
      "category": "agentic_computer_use",
      "benchmark_name": "Agents' Last Exam",
      "benchmark_version": null,
      "score_numeric": 26.3,
      "score_display": "26.3%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-08-26T00:00:00.000Z",
      "source": "https://z.ai/blog/glm-5.3-flash",
      "notes": null
    },
    {
      "id": "glm-5-3-flash-automationbench-48-8-none-none-none-z-ai",
      "model_slug": "glm-5-3-flash",
      "category": "agentic_computer_use",
      "benchmark_name": "AutomationBench",
      "benchmark_version": null,
      "score_numeric": 48.8,
      "score_display": "48.8",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": null,
      "source": "https://autoclaw.z.ai/blog/model/glm-5.3-flash/",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm53f-automation",
      "model_slug": "glm-5-3-flash",
      "category": "agentic_computer_use",
      "benchmark_name": "AutomationBench",
      "benchmark_version": "v1.0.6",
      "score_numeric": 48.8,
      "score_display": "48.8%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-08-26T00:00:00.000Z",
      "source": "https://z.ai/blog/glm-5.3-flash",
      "notes": null
    },
    {
      "id": "glm-5-3-flash-deepswe-1-1-63-4-none-none-none-z-ai",
      "model_slug": "glm-5-3-flash",
      "category": "coding",
      "benchmark_name": "DeepSWE",
      "benchmark_version": "1.1",
      "score_numeric": 63.4,
      "score_display": "63.4",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": null,
      "source": "https://autoclaw.z.ai/blog/model/glm-5.3-flash/",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm53f-deepswe",
      "model_slug": "glm-5-3-flash",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": null,
      "score_numeric": 63.4,
      "score_display": "63.4%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-08-26T00:00:00.000Z",
      "source": "https://z.ai/blog/glm-5.3-flash",
      "notes": null
    },
    {
      "id": "glm-5-3-flash-gdpval-aa-v2-1773-none-none-none-z-ai",
      "model_slug": "glm-5-3-flash",
      "category": "professional",
      "benchmark_name": "GDPval-AA",
      "benchmark_version": "v2",
      "score_numeric": 1773,
      "score_display": "1773",
      "score_unit": "Elo",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": null,
      "source": "https://autoclaw.z.ai/blog/model/glm-5.3-flash/",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm53f-gdp",
      "model_slug": "glm-5-3-flash",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA v2",
      "benchmark_version": null,
      "score_numeric": 1773,
      "score_display": "1773 Elo",
      "score_unit": "Elo",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Artificial Analysis",
      "evaluator": "Z.ai",
      "evaluation_date": "2026-08-26T00:00:00.000Z",
      "source": "https://z.ai/blog/glm-5.3-flash",
      "notes": "Artificial Analysis evaluation reported in Z.ai's launch table."
    },
    {
      "id": "glm-5-3-flash-humanity-s-last-exam-55-3-true-none-none-z-ai",
      "model_slug": "glm-5-3-flash",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 55.3,
      "score_display": "55.3",
      "score_unit": "%",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": null,
      "source": "https://autoclaw.z.ai/blog/model/glm-5.3-flash/",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm53f-hle",
      "model_slug": "glm-5-3-flash",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 55.3,
      "score_display": "55.3%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-08-26T00:00:00.000Z",
      "source": "https://z.ai/blog/glm-5.3-flash",
      "notes": null
    },
    {
      "id": "glm-5-3-flash-terminal-bench-2-1-84-3-none-none-none-z-ai",
      "model_slug": "glm-5-3-flash",
      "category": "coding",
      "benchmark_name": "Terminal-Bench",
      "benchmark_version": "2.1",
      "score_numeric": 84.3,
      "score_display": "84.3",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": null,
      "source": "https://autoclaw.z.ai/blog/model/glm-5.3-flash/",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm53f-term21",
      "model_slug": "glm-5-3-flash",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": null,
      "score_numeric": 84.3,
      "score_display": "84.3%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-08-26T00:00:00.000Z",
      "source": "https://z.ai/blog/glm-5.3-flash",
      "notes": null
    },
    {
      "id": "tq-20260907-031-glm53f-toolathlon",
      "model_slug": "glm-5-3-flash",
      "category": "agentic_computer_use",
      "benchmark_name": "Toolathlon",
      "benchmark_version": "Verified",
      "score_numeric": 78.4,
      "score_display": "78.4%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": "2026-08-26T00:00:00.000Z",
      "source": "https://z.ai/blog/glm-5.3-flash",
      "notes": null
    },
    {
      "id": "glm-5-3-flash-toolathlon-verified-78-4-none-none-none-z-ai",
      "model_slug": "glm-5-3-flash",
      "category": "agentic_computer_use",
      "benchmark_name": "Toolathlon Verified",
      "benchmark_version": null,
      "score_numeric": 78.4,
      "score_display": "78.4",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Z.ai",
      "evaluation_date": null,
      "source": "https://autoclaw.z.ai/blog/model/glm-5.3-flash/",
      "notes": null
    },
    {
      "id": "tq-20260907-034-9be88d51ebba7d6339da",
      "model_slug": "gpt-5-3-codex",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": null,
      "score_numeric": 77.3,
      "score_display": "77.3%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI GPT-5.4 comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-03-05T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-4/",
      "notes": null
    },
    {
      "id": "tq-20260907-034-ee90dfa46d4868fa7275",
      "model_slug": "gpt-5-3-codex",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld-Verified",
      "benchmark_version": null,
      "score_numeric": 74,
      "score_display": "74.0%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI revised evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-03-05T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-4/",
      "notes": "Supersedes the originally reported 64.7%; OpenAI reports 74.0% with the API parameter that preserves original image resolution."
    },
    {
      "id": "gpt-5-3-codex-swe-bench-pro-openai-comparison-eval-56-8-none-none-none-openai",
      "model_slug": "gpt-5-3-codex",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": "OpenAI comparison eval",
      "score_numeric": 56.8,
      "score_display": "56.8",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "OpenAI",
      "evaluation_date": null,
      "source": "https://openai.com/index/introducing-gpt-5-4/",
      "notes": null
    },
    {
      "id": "tq-20260907-034-8dccb771b67288fc3b10",
      "model_slug": "gpt-5-3-codex",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": "Public",
      "score_numeric": 56.8,
      "score_display": "56.8%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI launch evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-02-05T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-3-codex/",
      "notes": null
    },
    {
      "id": "tq-20260907-034-9e316d79132a5b60d4c5",
      "model_slug": "gpt-5-3-codex",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.0",
      "benchmark_version": null,
      "score_numeric": 77.3,
      "score_display": "77.3%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI launch evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-02-05T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-3-codex/",
      "notes": null
    },
    {
      "id": "tq-20260907-034-2f102cd44d5cf9a124b5",
      "model_slug": "gpt-5-3-codex",
      "category": "agentic_computer_use",
      "benchmark_name": "Toolathlon",
      "benchmark_version": null,
      "score_numeric": 51.9,
      "score_display": "51.9%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI GPT-5.4 comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-03-05T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-4/",
      "notes": null
    },
    {
      "id": "tq-20260907-034-d16b90e13bc9093f7f4d",
      "model_slug": "gpt-5-4",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": null,
      "score_numeric": 82.7,
      "score_display": "82.7%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "xhigh",
      "harness": "ChatGPT search tool",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-04-23T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-5/",
      "notes": null
    },
    {
      "id": "tq-20260907-034-753b2bab7dc66a962256",
      "model_slug": "gpt-5-4",
      "category": "math_reasoning",
      "benchmark_name": "FrontierMath",
      "benchmark_version": "Tier 1–3",
      "score_numeric": 47.6,
      "score_display": "47.6%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-04-23T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-5/",
      "notes": "Do not conflate with FrontierMath Tier 4 or FrontierMath v2."
    },
    {
      "id": "tq-20260907-034-815b719863df47941c32",
      "model_slug": "gpt-5-4",
      "category": "math_reasoning",
      "benchmark_name": "FrontierMath",
      "benchmark_version": "Tier 4",
      "score_numeric": 27.1,
      "score_display": "27.1%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-04-23T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-5/",
      "notes": "Do not conflate with Tier 1–3 or FrontierMath Tier 4 v2."
    },
    {
      "id": "tq-20260907-034-0ebf262822d6f61b4fbf",
      "model_slug": "gpt-5-4",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 92.8,
      "score_display": "92.8%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-04-23T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-5/",
      "notes": null
    },
    {
      "id": "tq-20260907-034-1af757697215a3b366d3",
      "model_slug": "gpt-5-4",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 39.8,
      "score_display": "39.8%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-04-23T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-5/",
      "notes": null
    },
    {
      "id": "tq-20260907-034-9433b629344e5b2d634b",
      "model_slug": "gpt-5-4",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 52.1,
      "score_display": "52.1%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-04-23T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-5/",
      "notes": null
    },
    {
      "id": "tq-20260907-034-b7f0af42550d30b11069",
      "model_slug": "gpt-5-4",
      "category": "agentic_computer_use",
      "benchmark_name": "MCP Atlas",
      "benchmark_version": "Public",
      "score_numeric": 70.6,
      "score_display": "70.6%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "xhigh",
      "harness": "Scale AI April 2026 update",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-04-23T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-5/",
      "notes": "Result reflects Scale AI's latest April 2026 benchmark update."
    },
    {
      "id": "gpt-5-4-osworld-verified-75-0-none-none-none-openai",
      "model_slug": "gpt-5-4",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld-Verified",
      "benchmark_version": null,
      "score_numeric": 75,
      "score_display": "75.0",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "OpenAI",
      "evaluation_date": null,
      "source": "https://openai.com/index/introducing-gpt-5-4/",
      "notes": null
    },
    {
      "id": "tq-20260907-034-9a7c7ab8124a0f954cbe",
      "model_slug": "gpt-5-4",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld-Verified",
      "benchmark_version": null,
      "score_numeric": 75,
      "score_display": "75.0%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-04-23T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-5/",
      "notes": null
    },
    {
      "id": "gpt-5-4-swe-bench-pro-57-7-none-none-none-openai",
      "model_slug": "gpt-5-4",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": null,
      "score_numeric": 57.7,
      "score_display": "57.7",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "OpenAI",
      "evaluation_date": null,
      "source": "https://openai.com/index/introducing-gpt-5-4/",
      "notes": null
    },
    {
      "id": "tq-20260907-034-2a1bdea4bdf9c7db94c9",
      "model_slug": "gpt-5-4",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": "Public",
      "score_numeric": 57.7,
      "score_display": "57.7%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-04-23T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-5/",
      "notes": "Latest OpenAI comparison table."
    },
    {
      "id": "gpt-5-4-terminal-bench-2-0-75-1-none-none-none-openai",
      "model_slug": "gpt-5-4",
      "category": "coding",
      "benchmark_name": "Terminal-Bench",
      "benchmark_version": "2.0",
      "score_numeric": 75.1,
      "score_display": "75.1",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "OpenAI",
      "evaluation_date": null,
      "source": "https://openai.com/index/introducing-gpt-5-4/",
      "notes": null
    },
    {
      "id": "tq-20260907-034-d500e022c8f3b61df60c",
      "model_slug": "gpt-5-4",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.0",
      "benchmark_version": null,
      "score_numeric": 75.1,
      "score_display": "75.1%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-04-23T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-5/",
      "notes": "Latest OpenAI comparison table."
    },
    {
      "id": "tq-20260907-034-b85c50123e00bd08dd9e",
      "model_slug": "gpt-5-4",
      "category": "agentic_computer_use",
      "benchmark_name": "Toolathlon",
      "benchmark_version": null,
      "score_numeric": 54.6,
      "score_display": "54.6%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-04-23T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-5/",
      "notes": null
    },
    {
      "id": "openai-20260317-gpt54mini-gpqa",
      "model_slug": "gpt-5-4-mini",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 88,
      "score_display": "88.0%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI GPT-5.4 mini/nano launch evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-03-17T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
      "notes": null
    },
    {
      "id": "openai-20260317-gpt54mini-osworld",
      "model_slug": "gpt-5-4-mini",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld",
      "benchmark_version": "Verified",
      "score_numeric": 72.1,
      "score_display": "72.1%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI GPT-5.4 mini/nano launch evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-03-17T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
      "notes": null
    },
    {
      "id": "openai-20260317-gpt54mini-swe-pro",
      "model_slug": "gpt-5-4-mini",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": "Public",
      "score_numeric": 54.4,
      "score_display": "54.4%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI GPT-5.4 mini/nano launch evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-03-17T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
      "notes": null
    },
    {
      "id": "openai-20260317-gpt54mini-terminal20",
      "model_slug": "gpt-5-4-mini",
      "category": "coding",
      "benchmark_name": "Terminal-Bench",
      "benchmark_version": "2.0",
      "score_numeric": 60,
      "score_display": "60.0%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI GPT-5.4 mini/nano launch evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-03-17T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
      "notes": null
    },
    {
      "id": "openai-20260317-gpt54mini-toolathlon",
      "model_slug": "gpt-5-4-mini",
      "category": "agentic_computer_use",
      "benchmark_name": "Toolathlon",
      "benchmark_version": null,
      "score_numeric": 42.9,
      "score_display": "42.9%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI GPT-5.4 mini/nano launch evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-03-17T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
      "notes": null
    },
    {
      "id": "openai-20260317-gpt54nano-gpqa",
      "model_slug": "gpt-5-4-nano",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 82.8,
      "score_display": "82.8%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI GPT-5.4 mini/nano launch evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-03-17T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
      "notes": null
    },
    {
      "id": "openai-20260317-gpt54nano-osworld",
      "model_slug": "gpt-5-4-nano",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld",
      "benchmark_version": "Verified",
      "score_numeric": 39,
      "score_display": "39.0%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI GPT-5.4 mini/nano launch evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-03-17T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
      "notes": null
    },
    {
      "id": "openai-20260317-gpt54nano-swe-pro",
      "model_slug": "gpt-5-4-nano",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": "Public",
      "score_numeric": 52.4,
      "score_display": "52.4%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI GPT-5.4 mini/nano launch evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-03-17T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
      "notes": null
    },
    {
      "id": "openai-20260317-gpt54nano-terminal20",
      "model_slug": "gpt-5-4-nano",
      "category": "coding",
      "benchmark_name": "Terminal-Bench",
      "benchmark_version": "2.0",
      "score_numeric": 46.3,
      "score_display": "46.3%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI GPT-5.4 mini/nano launch evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-03-17T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
      "notes": null
    },
    {
      "id": "openai-20260317-gpt54nano-toolathlon",
      "model_slug": "gpt-5-4-nano",
      "category": "agentic_computer_use",
      "benchmark_name": "Toolathlon",
      "benchmark_version": null,
      "score_numeric": 35.5,
      "score_display": "35.5%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI GPT-5.4 mini/nano launch evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-03-17T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
      "notes": null
    },
    {
      "id": "tq-20260907-034-d93dea8f63a59f995854",
      "model_slug": "gpt-5-4-pro",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": null,
      "score_numeric": 89.3,
      "score_display": "89.3%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "xhigh",
      "harness": "ChatGPT search tool",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-04-23T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-5/",
      "notes": null
    },
    {
      "id": "tq-20260907-034-0c1c62874a8696916d6f",
      "model_slug": "gpt-5-4-pro",
      "category": "math_reasoning",
      "benchmark_name": "FrontierMath",
      "benchmark_version": "Tier 1–3",
      "score_numeric": 50,
      "score_display": "50.0%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-04-23T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-5/",
      "notes": null
    },
    {
      "id": "tq-20260907-034-9c1bf6b54e4baae9da22",
      "model_slug": "gpt-5-4-pro",
      "category": "math_reasoning",
      "benchmark_name": "FrontierMath",
      "benchmark_version": "Tier 4",
      "score_numeric": 38,
      "score_display": "38.0%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-04-23T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-5/",
      "notes": null
    },
    {
      "id": "tq-20260907-034-7351af8109aa45c427de",
      "model_slug": "gpt-5-4-pro",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 94.4,
      "score_display": "94.4%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-04-23T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-5/",
      "notes": null
    },
    {
      "id": "tq-20260907-034-4681f7b6db504d32d112",
      "model_slug": "gpt-5-4-pro",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 42.7,
      "score_display": "42.7%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-04-23T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-5/",
      "notes": null
    },
    {
      "id": "tq-20260907-034-7e8d0039b12867bc0d37",
      "model_slug": "gpt-5-4-pro",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 58.7,
      "score_display": "58.7%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-04-23T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-5/",
      "notes": null
    },
    {
      "id": "openai-20260709-gpt55-browsecomp",
      "model_slug": "gpt-5-5",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": null,
      "score_numeric": 84.4,
      "score_display": "84.4%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": null
    },
    {
      "id": "openai-20260709-gpt55-deepswe",
      "model_slug": "gpt-5-5",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": "v1.1",
      "score_numeric": 67,
      "score_display": "67.0%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": null
    },
    {
      "id": "openai-20260709-gpt55-frontiermath13",
      "model_slug": "gpt-5-5",
      "category": "math_reasoning",
      "benchmark_name": "FrontierMath",
      "benchmark_version": "v2 Tier 1-3",
      "score_numeric": 85.3,
      "score_display": "85.3%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": null
    },
    {
      "id": "openai-20260709-gpt55-frontiermath4",
      "model_slug": "gpt-5-5",
      "category": "math_reasoning",
      "benchmark_name": "FrontierMath",
      "benchmark_version": "v2 Tier 4",
      "score_numeric": 72.5,
      "score_display": "72.5%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": null
    },
    {
      "id": "openai-20260709-gpt55-gdpvalaa2",
      "model_slug": "gpt-5-5",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA v2",
      "benchmark_version": "v2",
      "score_numeric": 1493.7,
      "score_display": "1,493.7 Elo",
      "score_unit": "Elo",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": null
    },
    {
      "id": "openai-20260709-gpt55-gpqa",
      "model_slug": "gpt-5-5",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 93.6,
      "score_display": "93.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": null
    },
    {
      "id": "openai-20260423-gpt55-hle-no-tools",
      "model_slug": "gpt-5-5",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 41.4,
      "score_display": "41.4%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI GPT-5.5 launch evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-04-23T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-5/",
      "notes": "No tools"
    },
    {
      "id": "openai-20260423-gpt55-hle-tools",
      "model_slug": "gpt-5-5",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 52.2,
      "score_display": "52.2%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI GPT-5.5 launch evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-04-23T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-5/",
      "notes": "With tools"
    },
    {
      "id": "openai-20260709-gpt55-osworld20",
      "model_slug": "gpt-5-5",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld 2.0",
      "benchmark_version": "2.0",
      "score_numeric": 47.5,
      "score_display": "47.5%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": null
    },
    {
      "id": "openai-20260709-gpt55-swe-pro",
      "model_slug": "gpt-5-5",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": null,
      "score_numeric": 59.4,
      "score_display": "59.4%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": "Later OpenAI comparison run; differs from the April GPT-5.5 launch score"
    },
    {
      "id": "openai-20260709-gpt55-terminal21",
      "model_slug": "gpt-5-5",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": "2.1",
      "score_numeric": 85.6,
      "score_display": "85.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": null
    },
    {
      "id": "tq-20260907-034-c7cb7cf1b58eaf889d2f",
      "model_slug": "gpt-5-5-pro",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": null,
      "score_numeric": 90.1,
      "score_display": "90.1%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "xhigh",
      "harness": "ChatGPT search tool",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-04-23T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-5/",
      "notes": null
    },
    {
      "id": "tq-20260907-034-2a25ccf9a4e53d9b982b",
      "model_slug": "gpt-5-5-pro",
      "category": "math_reasoning",
      "benchmark_name": "FrontierMath",
      "benchmark_version": "Tier 1–3",
      "score_numeric": 52.4,
      "score_display": "52.4%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-04-23T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-5/",
      "notes": null
    },
    {
      "id": "tq-20260907-034-987b4f96b8ed69894019",
      "model_slug": "gpt-5-5-pro",
      "category": "math_reasoning",
      "benchmark_name": "FrontierMath",
      "benchmark_version": "Tier 4",
      "score_numeric": 39.6,
      "score_display": "39.6%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-04-23T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-5/",
      "notes": null
    },
    {
      "id": "tq-20260907-034-384159ff77867fd2f9bf",
      "model_slug": "gpt-5-5-pro",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 57.2,
      "score_display": "57.2%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-04-23T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-5/",
      "notes": null
    },
    {
      "id": "tq-20260907-034-728cfbb1883b147a6cae",
      "model_slug": "gpt-5-5-pro",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 43.1,
      "score_display": "43.1%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "xhigh",
      "harness": "OpenAI evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-04-23T00:00:00.000Z",
      "source": "https://openai.com/index/introducing-gpt-5-5/",
      "notes": null
    },
    {
      "id": "openai-20260709-gpt56-luna-browsecomp",
      "model_slug": "gpt-5-6-luna",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": null,
      "score_numeric": 83.3,
      "score_display": "83.3%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": null
    },
    {
      "id": "openai-20260709-gpt56-luna-deepswe",
      "model_slug": "gpt-5-6-luna",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": "v1.1",
      "score_numeric": 67.2,
      "score_display": "67.2%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": null
    },
    {
      "id": "openai-20260709-gpt56-luna-frontiermath13",
      "model_slug": "gpt-5-6-luna",
      "category": "math_reasoning",
      "benchmark_name": "FrontierMath",
      "benchmark_version": "v2 Tier 1-3",
      "score_numeric": 78.6,
      "score_display": "78.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": null
    },
    {
      "id": "openai-20260709-gpt56-luna-frontiermath4",
      "model_slug": "gpt-5-6-luna",
      "category": "math_reasoning",
      "benchmark_name": "FrontierMath",
      "benchmark_version": "v2 Tier 4",
      "score_numeric": 58.5,
      "score_display": "58.5%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": null
    },
    {
      "id": "openai-20260709-gpt56-luna-gdpvalaa2",
      "model_slug": "gpt-5-6-luna",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA v2",
      "benchmark_version": "v2",
      "score_numeric": 1591.8,
      "score_display": "1,591.8 Elo",
      "score_unit": "Elo",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": null
    },
    {
      "id": "openai-20260709-gpt56-luna-gpqa",
      "model_slug": "gpt-5-6-luna",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 92.3,
      "score_display": "92.3%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": null
    },
    {
      "id": "openai-20260709-gpt56-luna-osworld20",
      "model_slug": "gpt-5-6-luna",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld 2.0",
      "benchmark_version": "2.0",
      "score_numeric": 45.6,
      "score_display": "45.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": null
    },
    {
      "id": "openai-20260709-gpt56-luna-swe-pro",
      "model_slug": "gpt-5-6-luna",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": null,
      "score_numeric": 62.7,
      "score_display": "62.7%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": "Public SWE-bench Pro result reported by OpenAI"
    },
    {
      "id": "openai-20260709-gpt56-luna-terminal21",
      "model_slug": "gpt-5-6-luna",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": "2.1",
      "score_numeric": 84.7,
      "score_display": "84.7%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": null
    },
    {
      "id": "openai-20260709-gpt56-sol-browsecomp",
      "model_slug": "gpt-5-6-sol",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": null,
      "score_numeric": 90.4,
      "score_display": "90.4%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Single-model GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": "Do not confuse with Sol Ultra / multi-agent 92.2%"
    },
    {
      "id": "openai-20260709-gpt56-sol-deepswe",
      "model_slug": "gpt-5-6-sol",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": "v1.1",
      "score_numeric": 72.7,
      "score_display": "72.7%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": null
    },
    {
      "id": "openai-20260709-gpt56-sol-frontiermath13",
      "model_slug": "gpt-5-6-sol",
      "category": "math_reasoning",
      "benchmark_name": "FrontierMath",
      "benchmark_version": "v2 Tier 1-3",
      "score_numeric": 89,
      "score_display": "89.0%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": null
    },
    {
      "id": "openai-20260709-gpt56-sol-frontiermath4",
      "model_slug": "gpt-5-6-sol",
      "category": "math_reasoning",
      "benchmark_name": "FrontierMath",
      "benchmark_version": "v2 Tier 4",
      "score_numeric": 83,
      "score_display": "83.0%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": null
    },
    {
      "id": "openai-20260709-gpt56-sol-gdpvalaa2",
      "model_slug": "gpt-5-6-sol",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA v2",
      "benchmark_version": "v2",
      "score_numeric": 1747.8,
      "score_display": "1,747.8 Elo",
      "score_unit": "Elo",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": null
    },
    {
      "id": "openai-20260709-gpt56-sol-gpqa",
      "model_slug": "gpt-5-6-sol",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 94.6,
      "score_display": "94.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": null
    },
    {
      "id": "openai-20260709-gpt56-sol-osworld20",
      "model_slug": "gpt-5-6-sol",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld 2.0",
      "benchmark_version": "2.0",
      "score_numeric": 62.6,
      "score_display": "62.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": null
    },
    {
      "id": "openai-20260709-gpt56-sol-swe-pro",
      "model_slug": "gpt-5-6-sol",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": null,
      "score_numeric": 64.6,
      "score_display": "64.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": "Public SWE-bench Pro result reported by OpenAI"
    },
    {
      "id": "openai-20260709-gpt56-sol-terminal21",
      "model_slug": "gpt-5-6-sol",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": "2.1",
      "score_numeric": 88.8,
      "score_display": "88.8%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": null
    },
    {
      "id": "openai-20260709-gpt56-terra-browsecomp",
      "model_slug": "gpt-5-6-terra",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": null,
      "score_numeric": 87.5,
      "score_display": "87.5%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": null
    },
    {
      "id": "openai-20260709-gpt56-terra-deepswe",
      "model_slug": "gpt-5-6-terra",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": "v1.1",
      "score_numeric": 69.6,
      "score_display": "69.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": null
    },
    {
      "id": "openai-20260709-gpt56-terra-frontiermath13",
      "model_slug": "gpt-5-6-terra",
      "category": "math_reasoning",
      "benchmark_name": "FrontierMath",
      "benchmark_version": "v2 Tier 1-3",
      "score_numeric": 84.9,
      "score_display": "84.9%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": null
    },
    {
      "id": "openai-20260709-gpt56-terra-frontiermath4",
      "model_slug": "gpt-5-6-terra",
      "category": "math_reasoning",
      "benchmark_name": "FrontierMath",
      "benchmark_version": "v2 Tier 4",
      "score_numeric": 68.3,
      "score_display": "68.3%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": null
    },
    {
      "id": "openai-20260709-gpt56-terra-gdpvalaa2",
      "model_slug": "gpt-5-6-terra",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA v2",
      "benchmark_version": "v2",
      "score_numeric": 1593,
      "score_display": "1,593 Elo",
      "score_unit": "Elo",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": null
    },
    {
      "id": "openai-20260709-gpt56-terra-gpqa",
      "model_slug": "gpt-5-6-terra",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 92.9,
      "score_display": "92.9%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": null
    },
    {
      "id": "openai-20260709-gpt56-terra-osworld20",
      "model_slug": "gpt-5-6-terra",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld 2.0",
      "benchmark_version": "2.0",
      "score_numeric": 50.2,
      "score_display": "50.2%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": null
    },
    {
      "id": "openai-20260709-gpt56-terra-swe-pro",
      "model_slug": "gpt-5-6-terra",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": null,
      "score_numeric": 63.4,
      "score_display": "63.4%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": "Public SWE-bench Pro result reported by OpenAI"
    },
    {
      "id": "openai-20260709-gpt56-terra-terminal21",
      "model_slug": "gpt-5-6-terra",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": "2.1",
      "score_numeric": 87.4,
      "score_display": "87.4%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenAI GPT-5.6 launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-5-6/",
      "notes": null
    },
    {
      "id": "gpt-6-astra-arc-agi-3-99-9-none-none-none-openai",
      "model_slug": "gpt-6-astra",
      "category": "math_reasoning",
      "benchmark_name": "ARC-AGI",
      "benchmark_version": "3",
      "score_numeric": 99.9,
      "score_display": "99.9",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "OpenAI",
      "evaluation_date": null,
      "source": "https://openai.com/index/gpt-6-astra/",
      "notes": null
    },
    {
      "id": "openai-20260903-astra-agents-last-exam",
      "model_slug": "gpt-6-astra",
      "category": "agentic_computer_use",
      "benchmark_name": "Agents' Last Exam",
      "benchmark_version": null,
      "score_numeric": 59.3,
      "score_display": "59.3%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "GPT-6 Astra launch evaluation",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-09-03T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-6-astra/",
      "notes": null
    },
    {
      "id": "openai-20260903-astra-deepswe-v1-1",
      "model_slug": "gpt-6-astra",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": null,
      "score_numeric": 74.1,
      "score_display": "74.1%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "GPT-6 Astra launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-09-03T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-6-astra/",
      "notes": "Cross-vendor comparison reported by OpenAI; preserve evaluator/harness when comparing."
    },
    {
      "id": "gpt-6-astra-exploitbench-100-none-none-none-openai",
      "model_slug": "gpt-6-astra",
      "category": "agentic_computer_use",
      "benchmark_name": "ExploitBench",
      "benchmark_version": null,
      "score_numeric": 100,
      "score_display": "100",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "OpenAI",
      "evaluation_date": null,
      "source": "https://openai.com/index/gpt-6-astra/",
      "notes": "Restricted-rollout evaluation; not a general consumer capability claim."
    },
    {
      "id": "openai-20260903-astra-frontiercode-1-1-extended",
      "model_slug": "gpt-6-astra",
      "category": "coding",
      "benchmark_name": "FrontierCode 1.1 Extended",
      "benchmark_version": null,
      "score_numeric": 64.5,
      "score_display": "64.5%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "GPT-6 Astra launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-09-03T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-6-astra/",
      "notes": "Cross-vendor comparison reported by OpenAI; preserve evaluator/harness when comparing."
    },
    {
      "id": "openai-20260903-astra-frontiercode-1-1-main",
      "model_slug": "gpt-6-astra",
      "category": "coding",
      "benchmark_name": "FrontierCode 1.1 Main",
      "benchmark_version": null,
      "score_numeric": 53.3,
      "score_display": "53.3%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "GPT-6 Astra launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-09-03T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-6-astra/",
      "notes": "Cross-vendor comparison reported by OpenAI; preserve evaluator/harness when comparing."
    },
    {
      "id": "gpt-6-astra-frontiermath-tier-4-98-none-none-none-openai",
      "model_slug": "gpt-6-astra",
      "category": "math_reasoning",
      "benchmark_name": "FrontierMath",
      "benchmark_version": "Tier 4",
      "score_numeric": 98,
      "score_display": "98",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "OpenAI",
      "evaluation_date": null,
      "source": "https://openai.com/index/gpt-6-astra/",
      "notes": null
    },
    {
      "id": "openai-20260903-astra-frontiermath-tier-4-v2",
      "model_slug": "gpt-6-astra",
      "category": "math_reasoning",
      "benchmark_name": "FrontierMath Tier 4 (v2)",
      "benchmark_version": null,
      "score_numeric": 97.6,
      "score_display": "97.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "GPT-6 Astra launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-09-03T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-6-astra/",
      "notes": "Cross-vendor comparison reported by OpenAI; preserve evaluator/harness when comparing."
    },
    {
      "id": "openai-20260903-astra-gpqa-diamond",
      "model_slug": "gpt-6-astra",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 96,
      "score_display": "96.0%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "GPT-6 Astra launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-09-03T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-6-astra/",
      "notes": "Cross-vendor comparison reported by OpenAI; preserve evaluator/harness when comparing."
    },
    {
      "id": "openai-20260903-astra-humanity-s-last-exam",
      "model_slug": "gpt-6-astra",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 57.2,
      "score_display": "57.2%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": "GPT-6 Astra launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-09-03T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-6-astra/",
      "notes": "Cross-vendor comparison reported by OpenAI; preserve evaluator/harness when comparing."
    },
    {
      "id": "openai-20260903-astra-terminal-bench-4-0",
      "model_slug": "gpt-6-astra",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 4.0",
      "benchmark_version": null,
      "score_numeric": 57.9,
      "score_display": "57.9%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "GPT-6 Astra launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-09-03T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-6-astra/",
      "notes": "Cross-vendor comparison reported by OpenAI; preserve evaluator/harness when comparing."
    },
    {
      "id": "openai-20260903-astra-terminal-bench-science-0-1",
      "model_slug": "gpt-6-astra",
      "category": "math_reasoning",
      "benchmark_name": "Terminal-Bench Science 0.1",
      "benchmark_version": null,
      "score_numeric": 64.6,
      "score_display": "64.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "GPT-6 Astra launch comparison suite",
      "evaluator": "OpenAI",
      "evaluation_date": "2026-09-03T00:00:00.000Z",
      "source": "https://openai.com/index/gpt-6-astra/",
      "notes": "Cross-vendor comparison reported by OpenAI; preserve evaluator/harness when comparing."
    },
    {
      "id": "tq-20260907-grok45-cursor",
      "model_slug": "grok-4-5",
      "category": "coding",
      "benchmark_name": "CursorBench",
      "benchmark_version": "3.2",
      "score_numeric": 66.7,
      "score_display": "66.7%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "high",
      "harness": "CursorBench 3.2",
      "evaluator": "SpaceXAI",
      "evaluation_date": "2026-08-12T00:00:00.000Z",
      "source": "https://x.ai/news/grok-4-6",
      "notes": null
    },
    {
      "id": "grok-4-5-deepswe-1-0-62-0-none-none-none-xai",
      "model_slug": "grok-4-5",
      "category": "coding",
      "benchmark_name": "DeepSWE",
      "benchmark_version": "1.0",
      "score_numeric": 62,
      "score_display": "62.0",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "xAI",
      "evaluation_date": null,
      "source": "https://x.ai/news/grok-4-5",
      "notes": null
    },
    {
      "id": "tq-20260907-grok45-deepswe",
      "model_slug": "grok-4-5",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": null,
      "score_numeric": 54,
      "score_display": "54.0%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "high",
      "harness": null,
      "evaluator": "SpaceXAI",
      "evaluation_date": "2026-08-12T00:00:00.000Z",
      "source": "https://x.ai/news/grok-4-6",
      "notes": "Later SpaceXAI comparison value; retained instead of the earlier 53% launch-table value."
    },
    {
      "id": "tq-20260907-grok45-frontiercode",
      "model_slug": "grok-4-5",
      "category": "coding",
      "benchmark_name": "FrontierCode 1.1 Extended",
      "benchmark_version": null,
      "score_numeric": 56.6,
      "score_display": "56.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "high",
      "harness": null,
      "evaluator": "SpaceXAI",
      "evaluation_date": "2026-08-12T00:00:00.000Z",
      "source": "https://x.ai/news/grok-4-6",
      "notes": null
    },
    {
      "id": "tq-20260907-grok45-gdpv2",
      "model_slug": "grok-4-5",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA v2",
      "benchmark_version": null,
      "score_numeric": 1526,
      "score_display": "1526 Elo",
      "score_unit": "Elo",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Artificial Analysis",
      "evaluator": "SpaceXAI",
      "evaluation_date": "2026-08-12T00:00:00.000Z",
      "source": "https://x.ai/news/grok-4-6",
      "notes": null
    },
    {
      "id": "grok-4-5-swe-marathon-29-0-none-none-none-xai",
      "model_slug": "grok-4-5",
      "category": "coding",
      "benchmark_name": "SWE Marathon",
      "benchmark_version": null,
      "score_numeric": 29,
      "score_display": "29.0",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "xAI",
      "evaluation_date": null,
      "source": "https://x.ai/news/grok-4-5",
      "notes": null
    },
    {
      "id": "grok-4-5-swe-bench-pro-64-7-none-none-none-xai",
      "model_slug": "grok-4-5",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": null,
      "score_numeric": 64.7,
      "score_display": "64.7",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "xAI",
      "evaluation_date": null,
      "source": "https://x.ai/news/grok-4-5",
      "notes": null
    },
    {
      "id": "tq-20260907-grok45-swepro",
      "model_slug": "grok-4-5",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": "Public",
      "score_numeric": 64.7,
      "score_display": "64.7%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "high",
      "harness": null,
      "evaluator": "SpaceXAI",
      "evaluation_date": "2026-07-16T00:00:00.000Z",
      "source": "https://x.ai/news/grok-4-5",
      "notes": null
    },
    {
      "id": "grok-4-5-terminal-bench-2-1-83-3-none-none-none-xai",
      "model_slug": "grok-4-5",
      "category": "coding",
      "benchmark_name": "Terminal-Bench",
      "benchmark_version": "2.1",
      "score_numeric": 83.3,
      "score_display": "83.3",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "xAI",
      "evaluation_date": null,
      "source": "https://x.ai/news/grok-4-5",
      "notes": null
    },
    {
      "id": "tq-20260907-grok45-terminal21",
      "model_slug": "grok-4-5",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": null,
      "score_numeric": 83.3,
      "score_display": "83.3%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "high",
      "harness": null,
      "evaluator": "SpaceXAI",
      "evaluation_date": "2026-07-16T00:00:00.000Z",
      "source": "https://x.ai/news/grok-4-5",
      "notes": null
    },
    {
      "id": "tq-20260907-grok45-terminal30",
      "model_slug": "grok-4-5",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 3.0",
      "benchmark_version": null,
      "score_numeric": 15.7,
      "score_display": "15.7%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "high",
      "harness": null,
      "evaluator": "SpaceXAI",
      "evaluation_date": "2026-08-12T00:00:00.000Z",
      "source": "https://x.ai/news/grok-4-6",
      "notes": null
    },
    {
      "id": "grok-4-6-apex-agents-57-5-none-none-none-xai",
      "model_slug": "grok-4-6",
      "category": "agentic_computer_use",
      "benchmark_name": "APEX-Agents",
      "benchmark_version": null,
      "score_numeric": 57.5,
      "score_display": "57.5",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "xAI",
      "evaluation_date": null,
      "source": "https://x.ai/news/grok-4-6",
      "notes": null
    },
    {
      "id": "grok-4-6-cursorbench-3-2-69-9-none-none-none-xai",
      "model_slug": "grok-4-6",
      "category": "coding",
      "benchmark_name": "CursorBench",
      "benchmark_version": "3.2",
      "score_numeric": 69.9,
      "score_display": "69.9",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "xAI",
      "evaluation_date": null,
      "source": "https://x.ai/news/grok-4-6",
      "notes": null
    },
    {
      "id": "tq-20260907-grok46-cursor",
      "model_slug": "grok-4-6",
      "category": "coding",
      "benchmark_name": "CursorBench",
      "benchmark_version": "3.2",
      "score_numeric": 69.9,
      "score_display": "69.9%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "high",
      "harness": "CursorBench 3.2",
      "evaluator": "SpaceXAI",
      "evaluation_date": "2026-08-12T00:00:00.000Z",
      "source": "https://x.ai/news/grok-4-6",
      "notes": null
    },
    {
      "id": "grok-4-6-deepswe-1-1-65-9-none-none-none-xai",
      "model_slug": "grok-4-6",
      "category": "coding",
      "benchmark_name": "DeepSWE",
      "benchmark_version": "1.1",
      "score_numeric": 65.9,
      "score_display": "65.9",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "xAI",
      "evaluation_date": null,
      "source": "https://x.ai/news/grok-4-6",
      "notes": null
    },
    {
      "id": "tq-20260907-grok46-deepswe",
      "model_slug": "grok-4-6",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": null,
      "score_numeric": 65.9,
      "score_display": "65.9%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "high",
      "harness": null,
      "evaluator": "SpaceXAI",
      "evaluation_date": "2026-08-12T00:00:00.000Z",
      "source": "https://x.ai/news/grok-4-6",
      "notes": null
    },
    {
      "id": "grok-4-6-frontiercode-1-1-extended-61-3-none-none-none-xai",
      "model_slug": "grok-4-6",
      "category": "coding",
      "benchmark_name": "FrontierCode",
      "benchmark_version": "1.1 Extended",
      "score_numeric": 61.3,
      "score_display": "61.3",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "xAI",
      "evaluation_date": null,
      "source": "https://x.ai/news/grok-4-6",
      "notes": null
    },
    {
      "id": "tq-20260907-grok46-frontiercode",
      "model_slug": "grok-4-6",
      "category": "coding",
      "benchmark_name": "FrontierCode 1.1 Extended",
      "benchmark_version": null,
      "score_numeric": 61.3,
      "score_display": "61.3%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "high",
      "harness": null,
      "evaluator": "SpaceXAI",
      "evaluation_date": "2026-08-12T00:00:00.000Z",
      "source": "https://x.ai/news/grok-4-6",
      "notes": null
    },
    {
      "id": "grok-4-6-gdpval-aa-v2-1753-none-none-none-xai",
      "model_slug": "grok-4-6",
      "category": "professional",
      "benchmark_name": "GDPval-AA",
      "benchmark_version": "v2",
      "score_numeric": 1753,
      "score_display": "1753",
      "score_unit": "Elo",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "xAI",
      "evaluation_date": null,
      "source": "https://x.ai/news/grok-4-6",
      "notes": null
    },
    {
      "id": "tq-20260907-grok46-gdpv2",
      "model_slug": "grok-4-6",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA v2",
      "benchmark_version": null,
      "score_numeric": 1753,
      "score_display": "1753 Elo",
      "score_unit": "Elo",
      "tools": true,
      "reasoning_effort": "high",
      "harness": "Artificial Analysis",
      "evaluator": "SpaceXAI",
      "evaluation_date": "2026-08-12T00:00:00.000Z",
      "source": "https://x.ai/news/grok-4-6",
      "notes": null
    },
    {
      "id": "tq-20260907-grok46-terminal30",
      "model_slug": "grok-4-6",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 3.0",
      "benchmark_version": null,
      "score_numeric": 26,
      "score_display": "26.0%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "high",
      "harness": null,
      "evaluator": "SpaceXAI",
      "evaluation_date": "2026-08-12T00:00:00.000Z",
      "source": "https://x.ai/news/grok-4-6",
      "notes": null
    },
    {
      "id": "tq-20260907-036-265385e9898b65e51cd8",
      "model_slug": "kimi-k2-5",
      "category": "math_reasoning",
      "benchmark_name": "AIME",
      "benchmark_version": "2025",
      "score_numeric": 96.1,
      "score_display": "96.1%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": "Kimi K2.5 official evaluation",
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-01-27T00:00:00.000Z",
      "source": "https://www.kimi.com/en/blog/kimi-k2-5",
      "notes": "avg@32."
    },
    {
      "id": "tq-20260907-036-3d508ad4377ba0592767",
      "model_slug": "kimi-k2-5",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": "context management",
      "score_numeric": 74.9,
      "score_display": "74.9%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "thinking",
      "harness": "Kimi K2.5 official evaluation",
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-01-27T00:00:00.000Z",
      "source": "https://www.kimi.com/en/blog/kimi-k2-5",
      "notes": "Context-managed condition."
    },
    {
      "id": "tq-20260907-036-7c36443d354928795205",
      "model_slug": "kimi-k2-5",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": "Agent Swarm",
      "score_numeric": 78.4,
      "score_display": "78.4%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "thinking",
      "harness": "Kimi K2.5 official evaluation",
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-01-27T00:00:00.000Z",
      "source": "https://www.kimi.com/en/blog/kimi-k2-5",
      "notes": "Agent Swarm: main agent max 15 steps; sub-agents max 100 steps."
    },
    {
      "id": "tq-20260907-036-ae0d94579a930dad0bb0",
      "model_slug": "kimi-k2-5",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": "discard-all",
      "score_numeric": 60.6,
      "score_display": "60.6%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "thinking",
      "harness": "Kimi K2.5 official evaluation",
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-01-27T00:00:00.000Z",
      "source": "https://www.kimi.com/en/blog/kimi-k2-5",
      "notes": "No context management beyond discard-all strategy."
    },
    {
      "id": "tq-20260907-036-53e8e17bac7a48f48f57",
      "model_slug": "kimi-k2-5",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 87.6,
      "score_display": "87.6%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": "Kimi K2.5 official evaluation",
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-01-27T00:00:00.000Z",
      "source": "https://www.kimi.com/en/blog/kimi-k2-5",
      "notes": "avg@8."
    },
    {
      "id": "kimi-k2-5-humanity-s-last-exam-multimodal-image-21-3-false-none-none-moonshot-ai",
      "model_slug": "kimi-k2-5",
      "category": "multimodal",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": "multimodal/image",
      "score_numeric": 21.3,
      "score_display": "21.3",
      "score_unit": "%",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Moonshot AI",
      "evaluation_date": null,
      "source": "https://www.kimi.com/en/blog/kimi-k2-5",
      "notes": null
    },
    {
      "id": "kimi-k2-5-humanity-s-last-exam-multimodal-image-39-8-true-none-none-moonshot-ai-tools",
      "model_slug": "kimi-k2-5",
      "category": "multimodal",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": "multimodal/image",
      "score_numeric": 39.8,
      "score_display": "39.8",
      "score_unit": "%",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Moonshot AI",
      "evaluation_date": null,
      "source": "https://www.kimi.com/en/blog/kimi-k2-5",
      "notes": null
    },
    {
      "id": "kimi-k2-5-humanity-s-last-exam-text-31-5-false-none-none-moonshot-ai",
      "model_slug": "kimi-k2-5",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": "text",
      "score_numeric": 31.5,
      "score_display": "31.5",
      "score_unit": "%",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Moonshot AI",
      "evaluation_date": null,
      "source": "https://www.kimi.com/en/blog/kimi-k2-5",
      "notes": null
    },
    {
      "id": "kimi-k2-5-humanity-s-last-exam-text-51-8-true-none-none-moonshot-ai-tools",
      "model_slug": "kimi-k2-5",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": "text",
      "score_numeric": 51.8,
      "score_display": "51.8",
      "score_unit": "%",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Moonshot AI",
      "evaluation_date": null,
      "source": "https://www.kimi.com/en/blog/kimi-k2-5",
      "notes": null
    },
    {
      "id": "tq-20260907-036-524c45e9a8f332ae524c",
      "model_slug": "kimi-k2-5",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": "Full multimodal",
      "score_numeric": 30.1,
      "score_display": "30.1%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": "Kimi K2.5 official evaluation",
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-01-27T00:00:00.000Z",
      "source": "https://www.kimi.com/en/blog/kimi-k2-5",
      "notes": "Full text+image set; no tools."
    },
    {
      "id": "tq-20260907-036-dc562bc8342a3f7f03b1",
      "model_slug": "kimi-k2-5",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": "Full multimodal",
      "score_numeric": 50.2,
      "score_display": "50.2%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "thinking",
      "harness": "Kimi K2.5 official evaluation",
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-01-27T00:00:00.000Z",
      "source": "https://www.kimi.com/en/blog/kimi-k2-5",
      "notes": "Full text+image set with search/code/browser tools."
    },
    {
      "id": "tq-20260907-036-3de6747cf488c1d4123a",
      "model_slug": "kimi-k2-5",
      "category": "coding",
      "benchmark_name": "LiveCodeBench",
      "benchmark_version": "v6",
      "score_numeric": 85,
      "score_display": "85%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": "Kimi K2.5 official evaluation",
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-01-27T00:00:00.000Z",
      "source": "https://www.kimi.com/en/blog/kimi-k2-5",
      "notes": null
    },
    {
      "id": "tq-20260907-036-91fc87cb9149e5b0dd27",
      "model_slug": "kimi-k2-5",
      "category": "knowledge",
      "benchmark_name": "MMLU-Pro",
      "benchmark_version": null,
      "score_numeric": 87.1,
      "score_display": "87.1%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": "Kimi K2.5 official evaluation",
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-01-27T00:00:00.000Z",
      "source": "https://www.kimi.com/en/blog/kimi-k2-5",
      "notes": null
    },
    {
      "id": "tq-20260907-036-d417197d9533aabd0c4d",
      "model_slug": "kimi-k2-5",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": "Public",
      "score_numeric": 50.7,
      "score_display": "50.7%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "non-thinking",
      "harness": "Kimi K2.5 official evaluation",
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-01-27T00:00:00.000Z",
      "source": "https://www.kimi.com/en/blog/kimi-k2-5",
      "notes": "Official Kimi framework; averaged over five runs."
    },
    {
      "id": "tq-20260907-036-92622f6be82109a4a508",
      "model_slug": "kimi-k2-5",
      "category": "coding",
      "benchmark_name": "SWE-bench Verified",
      "benchmark_version": null,
      "score_numeric": 76.8,
      "score_display": "76.8%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "non-thinking",
      "harness": "Kimi K2.5 official evaluation",
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-01-27T00:00:00.000Z",
      "source": "https://www.kimi.com/en/blog/kimi-k2-5",
      "notes": "Official Kimi framework; averaged over five runs."
    },
    {
      "id": "tq-20260907-036-5425f07043b9d3ebd831",
      "model_slug": "kimi-k2-5",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.0",
      "benchmark_version": null,
      "score_numeric": 50.8,
      "score_display": "50.8%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "non-thinking",
      "harness": "Kimi K2.5 official evaluation",
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-01-27T00:00:00.000Z",
      "source": "https://www.kimi.com/en/blog/kimi-k2-5",
      "notes": "Terminus-2 default agent; non-thinking because Kimi's thinking context management was incompatible with Terminus-2."
    },
    {
      "id": "tq-20260907-k26-aime",
      "model_slug": "kimi-k2-6",
      "category": "math_reasoning",
      "benchmark_name": "AIME",
      "benchmark_version": "2026",
      "score_numeric": 96.4,
      "score_display": "96.4%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-04-20T00:00:00.000Z",
      "source": "https://huggingface.co/moonshotai/Kimi-K2.6",
      "notes": null
    },
    {
      "id": "tq-20260907-k26-browsecomp",
      "model_slug": "kimi-k2-6",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": null,
      "score_numeric": 83.2,
      "score_display": "83.2%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "thinking",
      "harness": "Single-agent",
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-04-20T00:00:00.000Z",
      "source": "https://huggingface.co/moonshotai/Kimi-K2.6",
      "notes": "Agent Swarm result is intentionally not added as a second display value to avoid collapsing distinct harnesses in the UI."
    },
    {
      "id": "tq-20260907-k26-gpqa",
      "model_slug": "kimi-k2-6",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 90.5,
      "score_display": "90.5%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-04-20T00:00:00.000Z",
      "source": "https://huggingface.co/moonshotai/Kimi-K2.6",
      "notes": null
    },
    {
      "id": "tq-20260907-k26-hmmt",
      "model_slug": "kimi-k2-6",
      "category": "math_reasoning",
      "benchmark_name": "HMMT",
      "benchmark_version": "2026 Feb",
      "score_numeric": 92.7,
      "score_display": "92.7%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-04-20T00:00:00.000Z",
      "source": "https://huggingface.co/moonshotai/Kimi-K2.6",
      "notes": null
    },
    {
      "id": "tq-20260907-k26-hle-no-tools",
      "model_slug": "kimi-k2-6",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": "Full",
      "score_numeric": 34.7,
      "score_display": "34.7%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-04-20T00:00:00.000Z",
      "source": "https://huggingface.co/moonshotai/Kimi-K2.6",
      "notes": null
    },
    {
      "id": "tq-20260907-k26-hle-tools",
      "model_slug": "kimi-k2-6",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": "Full",
      "score_numeric": 54,
      "score_display": "54.0%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-04-20T00:00:00.000Z",
      "source": "https://huggingface.co/moonshotai/Kimi-K2.6",
      "notes": null
    },
    {
      "id": "tq-20260907-k26-livecode",
      "model_slug": "kimi-k2-6",
      "category": "coding",
      "benchmark_name": "LiveCodeBench",
      "benchmark_version": "v6",
      "score_numeric": 89.6,
      "score_display": "89.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-04-20T00:00:00.000Z",
      "source": "https://huggingface.co/moonshotai/Kimi-K2.6",
      "notes": null
    },
    {
      "id": "tq-20260907-k26-osworld",
      "model_slug": "kimi-k2-6",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld-Verified",
      "benchmark_version": null,
      "score_numeric": 73.1,
      "score_display": "73.1%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-04-20T00:00:00.000Z",
      "source": "https://huggingface.co/moonshotai/Kimi-K2.6",
      "notes": null
    },
    {
      "id": "tq-20260907-k26-swepro",
      "model_slug": "kimi-k2-6",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": "Public",
      "score_numeric": 58.6,
      "score_display": "58.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "thinking",
      "harness": "Moonshot in-house SWE-agent adaptation",
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-04-20T00:00:00.000Z",
      "source": "https://huggingface.co/moonshotai/Kimi-K2.6",
      "notes": "Average over 10 independent runs."
    },
    {
      "id": "tq-20260907-k26-sweverified",
      "model_slug": "kimi-k2-6",
      "category": "coding",
      "benchmark_name": "SWE-bench Verified",
      "benchmark_version": null,
      "score_numeric": 80.2,
      "score_display": "80.2%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "thinking",
      "harness": "Moonshot in-house SWE-agent adaptation",
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-04-20T00:00:00.000Z",
      "source": "https://huggingface.co/moonshotai/Kimi-K2.6",
      "notes": "Average over 10 independent runs."
    },
    {
      "id": "tq-20260907-k26-terminal20",
      "model_slug": "kimi-k2-6",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.0",
      "benchmark_version": "Terminus-2",
      "score_numeric": 66.7,
      "score_display": "66.7%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "thinking",
      "harness": "Terminus-2; preserve thinking",
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-04-20T00:00:00.000Z",
      "source": "https://huggingface.co/moonshotai/Kimi-K2.6",
      "notes": null
    },
    {
      "id": "tq-20260907-k26-toolathlon",
      "model_slug": "kimi-k2-6",
      "category": "agentic_computer_use",
      "benchmark_name": "Toolathlon",
      "benchmark_version": null,
      "score_numeric": 50,
      "score_display": "50.0%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-04-20T00:00:00.000Z",
      "source": "https://huggingface.co/moonshotai/Kimi-K2.6",
      "notes": null
    },
    {
      "id": "tq-20260907-kimi-k3-ale",
      "model_slug": "kimi-k3",
      "category": "agentic_computer_use",
      "benchmark_name": "Agents' Last Exam",
      "benchmark_version": null,
      "score_numeric": 28.3,
      "score_display": "28.3%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "Kimi K3 direct evaluation",
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-07-16T00:00:00.000Z",
      "source": "https://huggingface.co/moonshotai/Kimi-K3",
      "notes": null
    },
    {
      "id": "tq-20260907-kimi-k3-automation",
      "model_slug": "kimi-k3",
      "category": "agentic_computer_use",
      "benchmark_name": "AutomationBench",
      "benchmark_version": "Public 600-task subset",
      "score_numeric": 30.8,
      "score_display": "30.8%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "Kimi K3 direct evaluation",
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-07-16T00:00:00.000Z",
      "source": "https://huggingface.co/moonshotai/Kimi-K3",
      "notes": null
    },
    {
      "id": "tq-20260907-kimi-k3-browsecomp",
      "model_slug": "kimi-k3",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": null,
      "score_numeric": 91.2,
      "score_display": "91.2%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "Kimi agent with context compaction",
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-07-16T00:00:00.000Z",
      "source": "https://huggingface.co/moonshotai/Kimi-K3",
      "notes": "Kimi reports context compaction in the BrowseComp setup."
    },
    {
      "id": "tq-20260907-kimi-k3-deepswe",
      "model_slug": "kimi-k3",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": null,
      "score_numeric": 67.5,
      "score_display": "67.5%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "max",
      "harness": "Kimi Code",
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-07-16T00:00:00.000Z",
      "source": "https://huggingface.co/moonshotai/Kimi-K3",
      "notes": null
    },
    {
      "id": "tq-20260907-kimi-k3-gdpv2",
      "model_slug": "kimi-k3",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA v2",
      "benchmark_version": null,
      "score_numeric": 1686,
      "score_display": "1686 Elo",
      "score_unit": "Elo",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "Artificial Analysis",
      "evaluator": "Artificial Analysis",
      "evaluation_date": "2026-07-16T00:00:00.000Z",
      "source": "https://huggingface.co/moonshotai/Kimi-K3",
      "notes": "Value reproduced in Kimi's official comparison table."
    },
    {
      "id": "tq-20260907-kimi-k3-gpqa",
      "model_slug": "kimi-k3",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 93.5,
      "score_display": "93.5%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "max",
      "harness": "Kimi K3 direct evaluation",
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-07-16T00:00:00.000Z",
      "source": "https://huggingface.co/moonshotai/Kimi-K3",
      "notes": null
    },
    {
      "id": "tq-20260907-kimi-k3-hle-no-tools",
      "model_slug": "kimi-k3",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 43.5,
      "score_display": "43.5%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "max",
      "harness": "Kimi K3 direct evaluation",
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-07-16T00:00:00.000Z",
      "source": "https://huggingface.co/moonshotai/Kimi-K3",
      "notes": null
    },
    {
      "id": "tq-20260907-kimi-k3-hle-tools",
      "model_slug": "kimi-k3",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 56,
      "score_display": "56.0%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "Kimi K3 direct evaluation",
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-07-16T00:00:00.000Z",
      "source": "https://huggingface.co/moonshotai/Kimi-K3",
      "notes": null
    },
    {
      "id": "tq-20260907-kimi-k3-mcp-atlas",
      "model_slug": "kimi-k3",
      "category": "agentic_computer_use",
      "benchmark_name": "MCP Atlas",
      "benchmark_version": null,
      "score_numeric": 84.2,
      "score_display": "84.2%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "Kimi K3 direct evaluation",
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-07-16T00:00:00.000Z",
      "source": "https://huggingface.co/moonshotai/Kimi-K3",
      "notes": null
    },
    {
      "id": "tq-20260907-kimi-k3-osworld20",
      "model_slug": "kimi-k3",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld 2.0",
      "benchmark_version": null,
      "score_numeric": 58.3,
      "score_display": "58.3%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "Kimi K3 direct evaluation",
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-07-16T00:00:00.000Z",
      "source": "https://huggingface.co/moonshotai/Kimi-K3",
      "notes": null
    },
    {
      "id": "tq-20260907-kimi-k3-osworld-verified",
      "model_slug": "kimi-k3",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld-Verified",
      "benchmark_version": null,
      "score_numeric": 84.8,
      "score_display": "84.8%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "Kimi K3 direct evaluation",
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-07-16T00:00:00.000Z",
      "source": "https://huggingface.co/moonshotai/Kimi-K3",
      "notes": null
    },
    {
      "id": "tq-20260907-kimi-k3-terminal21",
      "model_slug": "kimi-k3",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": null,
      "score_numeric": 88.3,
      "score_display": "88.3%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "max",
      "harness": "Kimi Code",
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-07-16T00:00:00.000Z",
      "source": "https://huggingface.co/moonshotai/Kimi-K3",
      "notes": null
    },
    {
      "id": "tq-20260907-kimi-k3-toolathlon",
      "model_slug": "kimi-k3",
      "category": "agentic_computer_use",
      "benchmark_name": "Toolathlon",
      "benchmark_version": "Verified",
      "score_numeric": 76.5,
      "score_display": "76.5%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "Kimi K3 direct evaluation",
      "evaluator": "Moonshot AI",
      "evaluation_date": "2026-07-16T00:00:00.000Z",
      "source": "https://huggingface.co/moonshotai/Kimi-K3",
      "notes": null
    },
    {
      "id": "minimax-m2-5-browsecomp-76-3-none-none-with-context-management-minimax",
      "model_slug": "minimax-m2-5",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": null,
      "score_numeric": 76.3,
      "score_display": "76.3",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": "with context management",
      "evaluator": "MiniMax",
      "evaluation_date": null,
      "source": "https://www.minimax.io/news/minimax-m25",
      "notes": null
    },
    {
      "id": "tq-20260907-036-706052ee162ecca626cb",
      "model_slug": "minimax-m2-5",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": "context management",
      "score_numeric": 76.3,
      "score_display": "76.3%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": "MiniMax M2.5 launch evaluation",
      "evaluator": "MiniMax",
      "evaluation_date": "2026-02-12T00:00:00.000Z",
      "source": "https://www.minimax.io/news/minimax-m25",
      "notes": "MiniMax launch evaluation with context management."
    },
    {
      "id": "minimax-m2-5-multi-swe-bench-51-3-none-none-none-minimax",
      "model_slug": "minimax-m2-5",
      "category": "coding",
      "benchmark_name": "Multi-SWE-Bench",
      "benchmark_version": null,
      "score_numeric": 51.3,
      "score_display": "51.3",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "MiniMax",
      "evaluation_date": null,
      "source": "https://www.minimax.io/news/minimax-m25",
      "notes": null
    },
    {
      "id": "tq-20260907-036-4be892829af3f556694b",
      "model_slug": "minimax-m2-5",
      "category": "coding",
      "benchmark_name": "Multi-SWE-Bench",
      "benchmark_version": null,
      "score_numeric": 51.3,
      "score_display": "51.3%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "MiniMax M2.5 launch evaluation",
      "evaluator": "MiniMax",
      "evaluation_date": "2026-02-12T00:00:00.000Z",
      "source": "https://www.minimax.io/news/minimax-m25",
      "notes": null
    },
    {
      "id": "minimax-m2-5-swe-bench-verified-80-2-none-none-none-minimax",
      "model_slug": "minimax-m2-5",
      "category": "coding",
      "benchmark_name": "SWE-bench Verified",
      "benchmark_version": null,
      "score_numeric": 80.2,
      "score_display": "80.2",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "MiniMax",
      "evaluation_date": null,
      "source": "https://www.minimax.io/news/minimax-m25",
      "notes": null
    },
    {
      "id": "tq-20260907-036-537bcca44ddcc3caf306",
      "model_slug": "minimax-m2-5",
      "category": "coding",
      "benchmark_name": "SWE-bench Verified",
      "benchmark_version": null,
      "score_numeric": 80.2,
      "score_display": "80.2%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "MiniMax M2.5 launch evaluation",
      "evaluator": "MiniMax",
      "evaluation_date": "2026-02-12T00:00:00.000Z",
      "source": "https://www.minimax.io/news/minimax-m25",
      "notes": null
    },
    {
      "id": "minimax-m2-7-gdpval-aa-1495-none-none-none-minimax",
      "model_slug": "minimax-m2-7",
      "category": "professional",
      "benchmark_name": "GDPval-AA",
      "benchmark_version": null,
      "score_numeric": 1495,
      "score_display": "1495",
      "score_unit": "Elo",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "MiniMax",
      "evaluation_date": null,
      "source": "https://www.minimax.io/news/minimax-m27-en",
      "notes": null
    },
    {
      "id": "tq-20260907-036-63c7faf9cdb4143a6fce",
      "model_slug": "minimax-m2-7",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA",
      "benchmark_version": null,
      "score_numeric": 1495,
      "score_display": "1495 Elo",
      "score_unit": "elo",
      "tools": null,
      "reasoning_effort": null,
      "harness": "MiniMax M2.7 launch evaluation",
      "evaluator": "MiniMax",
      "evaluation_date": "2026-03-18T00:00:00.000Z",
      "source": "https://www.minimax.io/news/minimax-m27-en",
      "notes": "Official MiniMax launch labels this GDPval-AA; do not relabel as GDPval-AA v2."
    },
    {
      "id": "minimax-m2-7-swe-bench-pro-56-22-none-none-none-minimax",
      "model_slug": "minimax-m2-7",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": null,
      "score_numeric": 56.22,
      "score_display": "56.22",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "MiniMax",
      "evaluation_date": null,
      "source": "https://www.minimax.io/news/minimax-m27-en",
      "notes": null
    },
    {
      "id": "tq-20260907-036-69ef487a335b958ff2af",
      "model_slug": "minimax-m2-7",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": "Public",
      "score_numeric": 56.22,
      "score_display": "56.22%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "MiniMax M2.7 launch evaluation",
      "evaluator": "MiniMax",
      "evaluation_date": "2026-03-18T00:00:00.000Z",
      "source": "https://www.minimax.io/news/minimax-m27-en",
      "notes": null
    },
    {
      "id": "minimax-m2-7-terminal-bench-2-57-0-none-none-none-minimax",
      "model_slug": "minimax-m2-7",
      "category": "coding",
      "benchmark_name": "Terminal-Bench",
      "benchmark_version": "2",
      "score_numeric": 57,
      "score_display": "57.0",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "MiniMax",
      "evaluation_date": null,
      "source": "https://www.minimax.io/news/minimax-m27-en",
      "notes": null
    },
    {
      "id": "tq-20260907-036-4886bf25070e581d6c93",
      "model_slug": "minimax-m2-7",
      "category": "coding",
      "benchmark_name": "Terminal-Bench",
      "benchmark_version": "2",
      "score_numeric": 57,
      "score_display": "57%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "MiniMax M2.7 launch evaluation",
      "evaluator": "MiniMax",
      "evaluation_date": "2026-03-18T00:00:00.000Z",
      "source": "https://www.minimax.io/news/minimax-m27-en",
      "notes": null
    },
    {
      "id": "minimax-m2-7-vibe-pro-55-6-none-none-none-minimax",
      "model_slug": "minimax-m2-7",
      "category": "coding",
      "benchmark_name": "VIBE-Pro",
      "benchmark_version": null,
      "score_numeric": 55.6,
      "score_display": "55.6",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "MiniMax",
      "evaluation_date": null,
      "source": "https://www.minimax.io/news/minimax-m27-en",
      "notes": null
    },
    {
      "id": "tq-20260907-036-8613bcc670d70df3edfa",
      "model_slug": "minimax-m2-7",
      "category": "coding",
      "benchmark_name": "VIBE-Pro",
      "benchmark_version": null,
      "score_numeric": 55.6,
      "score_display": "55.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "MiniMax M2.7 launch evaluation",
      "evaluator": "MiniMax",
      "evaluation_date": "2026-03-18T00:00:00.000Z",
      "source": "https://www.minimax.io/news/minimax-m27-en",
      "notes": null
    },
    {
      "id": "minimax-m3-browsecomp-83-5-none-none-none-minimax",
      "model_slug": "minimax-m3",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": null,
      "score_numeric": 83.5,
      "score_display": "83.5",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "MiniMax",
      "evaluation_date": null,
      "source": "https://www.minimax.io/blog/minimax-m3",
      "notes": null
    },
    {
      "id": "tq-20260907-minimax-m3-browsecomp",
      "model_slug": "minimax-m3",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": null,
      "score_numeric": 83.5,
      "score_display": "83.5%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": "WebExplorer agent framework",
      "evaluator": "MiniMax",
      "evaluation_date": "2026-06-01T00:00:00.000Z",
      "source": "https://www.minimax.io/news/minimax-m3",
      "notes": "Evaluation discards history beyond 64K tokens."
    },
    {
      "id": "tq-20260907-minimax-m3-mcp-atlas",
      "model_slug": "minimax-m3",
      "category": "agentic_computer_use",
      "benchmark_name": "MCP Atlas",
      "benchmark_version": "Public set",
      "score_numeric": 74.2,
      "score_display": "74.2%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Official MCP Atlas codebase",
      "evaluator": "MiniMax",
      "evaluation_date": "2026-06-01T00:00:00.000Z",
      "source": "https://www.minimax.io/news/minimax-m3",
      "notes": null
    },
    {
      "id": "tq-20260907-minimax-m3-swe-pro",
      "model_slug": "minimax-m3",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": null,
      "score_numeric": 59,
      "score_display": "59.0%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Claude Code scaffold on MiniMax internal infrastructure",
      "evaluator": "MiniMax",
      "evaluation_date": "2026-06-01T00:00:00.000Z",
      "source": "https://www.minimax.io/news/minimax-m3",
      "notes": null
    },
    {
      "id": "tq-20260907-minimax-m3-terminal21",
      "model_slug": "minimax-m3",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": null,
      "score_numeric": 66,
      "score_display": "66.0%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Terminus 2; 8C16G; 2h",
      "evaluator": "MiniMax",
      "evaluation_date": "2026-06-01T00:00:00.000Z",
      "source": "https://www.minimax.io/news/minimax-m3",
      "notes": "MiniMax reports a 128K max output setting for this evaluation harness; that is not promoted here as a universal model output limit."
    },
    {
      "id": "mistral-medium-3-5-swe-bench-verified-77-6-none-none-none-mistral-ai",
      "model_slug": "mistral-medium-3-5",
      "category": "coding",
      "benchmark_name": "SWE-bench Verified",
      "benchmark_version": null,
      "score_numeric": 77.6,
      "score_display": "77.6",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Mistral AI",
      "evaluation_date": null,
      "source": "https://docs.mistral.ai/models/mistral-medium-3-5-26-04",
      "notes": null
    },
    {
      "id": "tq-20260907-030-mistral-medium-3-5-swev",
      "model_slug": "mistral-medium-3-5",
      "category": "coding",
      "benchmark_name": "SWE-bench Verified",
      "benchmark_version": null,
      "score_numeric": 77.6,
      "score_display": "77.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Mistral AI",
      "evaluation_date": "2026-04-28T00:00:00.000Z",
      "source": "https://mistral.ai/news/vibe-remote-agents-mistral-medium-3-5/",
      "notes": null
    },
    {
      "id": "mistral-medium-3-5-τ³-telecom-91-4-none-none-none-mistral-ai",
      "model_slug": "mistral-medium-3-5",
      "category": "agentic_computer_use",
      "benchmark_name": "τ³-Telecom",
      "benchmark_version": null,
      "score_numeric": 91.4,
      "score_display": "91.4",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Mistral AI",
      "evaluation_date": null,
      "source": "https://docs.mistral.ai/models/mistral-medium-3-5-26-04",
      "notes": null
    },
    {
      "id": "tq-20260907-030-mistral-medium-3-5-tau3",
      "model_slug": "mistral-medium-3-5",
      "category": "agentic_computer_use",
      "benchmark_name": "τ³-Telecom",
      "benchmark_version": null,
      "score_numeric": 91.4,
      "score_display": "91.4%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Mistral AI",
      "evaluation_date": "2026-04-28T00:00:00.000Z",
      "source": "https://mistral.ai/news/vibe-remote-agents-mistral-medium-3-5/",
      "notes": "First-party agentic score; benchmark is retained normalized even though it is not in the comparison template."
    },
    {
      "id": "tq-20260907-030-mistral-small-4-aalcr",
      "model_slug": "mistral-small-4",
      "category": "other",
      "benchmark_name": "AA-LCR",
      "benchmark_version": null,
      "score_numeric": 0.72,
      "score_display": "0.72",
      "score_unit": "score",
      "tools": null,
      "reasoning_effort": "reasoning",
      "harness": null,
      "evaluator": "Mistral AI",
      "evaluation_date": "2026-03-16T00:00:00.000Z",
      "source": "https://mistral.ai/news/mistral-small-4/",
      "notes": "First-party score; non-canonical comparison row retained as normalized evidence."
    },
    {
      "id": "tq-20260906-benchfill-glimmer-aime2026",
      "model_slug": "muse-glimmer",
      "category": "math_reasoning",
      "benchmark_name": "AIME",
      "benchmark_version": "2026",
      "score_numeric": 94.7,
      "score_display": "94.7%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "High",
      "harness": "Muse Glimmer high-reasoning evaluation",
      "evaluator": "Meta",
      "evaluation_date": "2026-08-10T00:00:00.000Z",
      "source": "https://huggingface.co/meta-models/Muse-Glimmer-30B",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-glimmer-gdpv2",
      "model_slug": "muse-glimmer",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA v2",
      "benchmark_version": null,
      "score_numeric": 953,
      "score_display": "953 Elo",
      "score_unit": "Elo",
      "tools": true,
      "reasoning_effort": "High",
      "harness": "Stirrup",
      "evaluator": "Artificial Analysis",
      "evaluation_date": "2026-08-10T00:00:00.000Z",
      "source": "https://huggingface.co/meta-models/Muse-Glimmer-30B",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-glimmer-gpqa",
      "model_slug": "muse-glimmer",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 83.5,
      "score_display": "83.5%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "High",
      "harness": "AA evaluation",
      "evaluator": "Artificial Analysis",
      "evaluation_date": "2026-08-10T00:00:00.000Z",
      "source": "https://huggingface.co/meta-models/Muse-Glimmer-30B",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-glimmer-hle",
      "model_slug": "muse-glimmer",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 22,
      "score_display": "22.0%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "High",
      "harness": "Text evaluation",
      "evaluator": "Artificial Analysis",
      "evaluation_date": "2026-08-10T00:00:00.000Z",
      "source": "https://huggingface.co/meta-models/Muse-Glimmer-30B",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-glimmer-mcp-atlas",
      "model_slug": "muse-glimmer",
      "category": "agentic_computer_use",
      "benchmark_name": "MCP Atlas",
      "benchmark_version": "Public",
      "score_numeric": 75.5,
      "score_display": "75.5%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "High",
      "harness": "Muse Glimmer high-reasoning evaluation",
      "evaluator": "Meta",
      "evaluation_date": "2026-08-10T00:00:00.000Z",
      "source": "https://huggingface.co/meta-models/Muse-Glimmer-30B",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-glimmer-osworld-verified",
      "model_slug": "muse-glimmer",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld-Verified",
      "benchmark_version": null,
      "score_numeric": 65.9,
      "score_display": "65.9%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "High",
      "harness": "Muse Glimmer high-reasoning evaluation",
      "evaluator": "Meta",
      "evaluation_date": "2026-08-10T00:00:00.000Z",
      "source": "https://huggingface.co/meta-models/Muse-Glimmer-30B",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-glimmer-swe-pro",
      "model_slug": "muse-glimmer",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": "Public",
      "score_numeric": 51.2,
      "score_display": "51.2%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "High",
      "harness": "ScaleAI SWE-bench Pro",
      "evaluator": "Meta",
      "evaluation_date": "2026-08-10T00:00:00.000Z",
      "source": "https://huggingface.co/meta-models/Muse-Glimmer-30B",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-glimmer-swe-verified",
      "model_slug": "muse-glimmer",
      "category": "coding",
      "benchmark_name": "SWE-bench Verified",
      "benchmark_version": null,
      "score_numeric": 76,
      "score_display": "76.0%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "High",
      "harness": "SWE-bench Verified",
      "evaluator": "Meta",
      "evaluation_date": "2026-08-10T00:00:00.000Z",
      "source": "https://huggingface.co/meta-models/Muse-Glimmer-30B",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-glimmer-terminal21",
      "model_slug": "muse-glimmer",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": "Terminus-2",
      "score_numeric": 51.7,
      "score_display": "51.7%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "High",
      "harness": "Terminus-2",
      "evaluator": "Artificial Analysis",
      "evaluation_date": "2026-08-10T00:00:00.000Z",
      "source": "https://huggingface.co/meta-models/Muse-Glimmer-30B",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-muse1-deepswe",
      "model_slug": "muse-spark",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": null,
      "score_numeric": 10,
      "score_display": "10.0%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "Contemplating",
      "harness": "Muse Spark 1.1 launch comparison",
      "evaluator": "Meta",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://research.meta.ai/blog/introducing-muse-spark-meta-model-api",
      "notes": "Comparator result reported by Meta in the Muse Spark 1.1 evaluation."
    },
    {
      "id": "muse-spark-frontierscience-research-38-none-none-none-meta",
      "model_slug": "muse-spark",
      "category": "knowledge",
      "benchmark_name": "FrontierScience Research",
      "benchmark_version": null,
      "score_numeric": 38,
      "score_display": "38",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Meta",
      "evaluation_date": null,
      "source": "https://ai.meta.com/blog/introducing-muse-spark-msl/",
      "notes": null
    },
    {
      "id": "meta-20260408-muse-spark-hle-contemplating",
      "model_slug": "muse-spark",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 58,
      "score_display": "58%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": "Contemplating",
      "harness": "Muse Spark launch evaluation",
      "evaluator": "Meta",
      "evaluation_date": "2026-04-08T00:00:00.000Z",
      "source": "https://ai.meta.com/blog/introducing-muse-spark-msl/",
      "notes": "Contemplating mode uses parallel multi-agent test-time reasoning."
    },
    {
      "id": "muse-spark-humanity-s-last-exam-58-none-contemplating-none-meta",
      "model_slug": "muse-spark",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 58,
      "score_display": "58",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": "Contemplating",
      "harness": null,
      "evaluator": "Meta",
      "evaluation_date": null,
      "source": "https://ai.meta.com/blog/introducing-muse-spark-msl/",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-muse1-hle-tools",
      "model_slug": "muse-spark",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 50.4,
      "score_display": "50.4%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "Contemplating",
      "harness": "Muse Spark 1.1 launch comparison",
      "evaluator": "Meta",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://research.meta.ai/blog/introducing-muse-spark-meta-model-api",
      "notes": "Comparator result reported by Meta in the Muse Spark 1.1 evaluation."
    },
    {
      "id": "tq-20260906-benchfill-muse1-mcp-atlas",
      "model_slug": "muse-spark",
      "category": "agentic_computer_use",
      "benchmark_name": "MCP Atlas",
      "benchmark_version": null,
      "score_numeric": 82.2,
      "score_display": "82.2%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "Contemplating",
      "harness": "Muse Spark 1.1 launch comparison",
      "evaluator": "Meta",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://research.meta.ai/blog/introducing-muse-spark-meta-model-api",
      "notes": "Comparator result reported by Meta in the Muse Spark 1.1 evaluation."
    },
    {
      "id": "tq-20260906-benchfill-muse1-osworld-verified",
      "model_slug": "muse-spark",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld-Verified",
      "benchmark_version": null,
      "score_numeric": 53.3,
      "score_display": "53.3%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "Contemplating",
      "harness": "Muse Spark 1.1 launch comparison",
      "evaluator": "Meta",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://research.meta.ai/blog/introducing-muse-spark-meta-model-api",
      "notes": "Comparator result reported by Meta in the Muse Spark 1.1 evaluation."
    },
    {
      "id": "tq-20260906-benchfill-muse1-swe-pro",
      "model_slug": "muse-spark",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": "Public",
      "score_numeric": 55,
      "score_display": "55.0%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "Contemplating",
      "harness": "Muse Spark 1.1 launch comparison",
      "evaluator": "Meta",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://research.meta.ai/blog/introducing-muse-spark-meta-model-api",
      "notes": "Comparator result reported by Meta in the Muse Spark 1.1 evaluation."
    },
    {
      "id": "tq-20260906-benchfill-muse1-terminal21",
      "model_slug": "muse-spark",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": null,
      "score_numeric": 67.3,
      "score_display": "67.3%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "Contemplating",
      "harness": "Muse Spark 1.1 launch comparison",
      "evaluator": "Meta",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://research.meta.ai/blog/introducing-muse-spark-meta-model-api",
      "notes": "Comparator result reported by Meta in the Muse Spark 1.1 evaluation."
    },
    {
      "id": "tq-20260906-benchfill-muse1-toolathlon",
      "model_slug": "muse-spark",
      "category": "agentic_computer_use",
      "benchmark_name": "Toolathlon",
      "benchmark_version": "Verified",
      "score_numeric": 49.4,
      "score_display": "49.4%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "Contemplating",
      "harness": "Muse Spark 1.1 launch comparison",
      "evaluator": "Meta",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://research.meta.ai/blog/introducing-muse-spark-meta-model-api",
      "notes": "Comparator result reported by Meta in the Muse Spark 1.1 evaluation."
    },
    {
      "id": "tq-20260906-benchfill-muse11-deepswe",
      "model_slug": "muse-spark-1-1",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": null,
      "score_numeric": 53.3,
      "score_display": "53.3%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "xhigh",
      "harness": "mini-swe-agent fork",
      "evaluator": "Meta",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://research.meta.ai/blog/introducing-muse-spark-meta-model-api",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-muse11-gdpv2",
      "model_slug": "muse-spark-1-1",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA v2",
      "benchmark_version": null,
      "score_numeric": 1381,
      "score_display": "1381 Elo",
      "score_unit": "Elo",
      "tools": true,
      "reasoning_effort": "xhigh",
      "harness": "Stirrup",
      "evaluator": "Artificial Analysis",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://research.meta.ai/blog/introducing-muse-spark-meta-model-api",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-muse11-hle-tools",
      "model_slug": "muse-spark-1-1",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 62.1,
      "score_display": "62.1%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "xhigh",
      "harness": "Browser + bash tools",
      "evaluator": "Meta",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://research.meta.ai/blog/introducing-muse-spark-meta-model-api",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-muse11-mcp-atlas",
      "model_slug": "muse-spark-1-1",
      "category": "agentic_computer_use",
      "benchmark_name": "MCP Atlas",
      "benchmark_version": null,
      "score_numeric": 88.1,
      "score_display": "88.1%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "xhigh",
      "harness": "Muse Spark 1.1 evaluation",
      "evaluator": "Meta",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://research.meta.ai/blog/introducing-muse-spark-meta-model-api",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-muse11-osworld-verified",
      "model_slug": "muse-spark-1-1",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld-Verified",
      "benchmark_version": null,
      "score_numeric": 80.8,
      "score_display": "80.8%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "xhigh",
      "harness": "Muse Spark 1.1 evaluation",
      "evaluator": "Meta",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://research.meta.ai/blog/introducing-muse-spark-meta-model-api",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-muse11-swe-pro",
      "model_slug": "muse-spark-1-1",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": "Public",
      "score_numeric": 61.5,
      "score_display": "61.5%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "xhigh",
      "harness": "mini-swe-agent",
      "evaluator": "Scale AI",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://research.meta.ai/blog/introducing-muse-spark-meta-model-api",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-muse11-terminal21",
      "model_slug": "muse-spark-1-1",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": null,
      "score_numeric": 80,
      "score_display": "80.0%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "xhigh",
      "harness": "Muse Spark 1.1 agent harness",
      "evaluator": "Meta",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://research.meta.ai/blog/introducing-muse-spark-meta-model-api",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-muse11-toolathlon",
      "model_slug": "muse-spark-1-1",
      "category": "agentic_computer_use",
      "benchmark_name": "Toolathlon",
      "benchmark_version": "Verified",
      "score_numeric": 75.6,
      "score_display": "75.6%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "xhigh",
      "harness": "Toolathlon-Verified",
      "evaluator": "Meta",
      "evaluation_date": "2026-07-09T00:00:00.000Z",
      "source": "https://research.meta.ai/blog/introducing-muse-spark-meta-model-api",
      "notes": null
    },
    {
      "id": "meta-20260902-muse12-agentic-if",
      "model_slug": "muse-spark-1-2",
      "category": "agentic_computer_use",
      "benchmark_name": "Agentic IF Index",
      "benchmark_version": "Meta internal",
      "score_numeric": 46.2,
      "score_display": "46.2%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "xhigh",
      "harness": "Meta internal agentic instruction-following evaluation",
      "evaluator": "Meta",
      "evaluation_date": "2026-09-02T00:00:00.000Z",
      "source": "https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology",
      "notes": "Internal benchmark; shown for same-vendor generation comparison, not as an independent leaderboard."
    },
    {
      "id": "meta-20260902-muse12-automation",
      "model_slug": "muse-spark-1-2",
      "category": "agentic_computer_use",
      "benchmark_name": "AutomationBench",
      "benchmark_version": null,
      "score_numeric": 38.2,
      "score_display": "38.2%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "xhigh",
      "harness": "Muse Spark 1.3 launch scorecard",
      "evaluator": "Meta",
      "evaluation_date": "2026-09-02T00:00:00.000Z",
      "source": "https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology",
      "notes": "1.2 baseline published in Meta's 1.3 evaluation table."
    },
    {
      "id": "gdm-20260813-muse12-deepswe",
      "model_slug": "muse-spark-1-2",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": null,
      "score_numeric": 54.9,
      "score_display": "54.9%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Gemini 3.7 Flash model-card comparison",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-08-13T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
      "notes": "Cross-vendor score reported by Google DeepMind; preserve evaluator and harness."
    },
    {
      "id": "tq-20260906-benchfill-muse12-deepswe-meta",
      "model_slug": "muse-spark-1-2",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": null,
      "score_numeric": 59.3,
      "score_display": "59.3%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "xhigh",
      "harness": "Muse Code",
      "evaluator": "Meta",
      "evaluation_date": "2026-08-05T00:00:00.000Z",
      "source": "https://research.meta.ai/blog/introducing-muse-code-and-muse-spark-1-2",
      "notes": "Meta first-party evaluation; retained separately from Google's cross-vendor comparison row."
    },
    {
      "id": "meta-20260902-muse12-deepsearchqa",
      "model_slug": "muse-spark-1-2",
      "category": "agentic_computer_use",
      "benchmark_name": "DeepSearchQA",
      "benchmark_version": null,
      "score_numeric": 85.9,
      "score_display": "85.9%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "xhigh",
      "harness": "Muse Spark 1.3 launch scorecard",
      "evaluator": "Meta",
      "evaluation_date": "2026-09-02T00:00:00.000Z",
      "source": "https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology",
      "notes": "1.2 baseline published in Meta's 1.3 evaluation table."
    },
    {
      "id": "gdm-20260813-muse12-gdpv2",
      "model_slug": "muse-spark-1-2",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA v2",
      "benchmark_version": null,
      "score_numeric": 1628,
      "score_display": "1628 Elo",
      "score_unit": "Elo",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Artificial Analysis / Gemini 3.7 model-card comparison",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-08-13T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
      "notes": "Cross-vendor table value; GDPval-AA v2 must not be compared as if it were the older GDPval-AA variant."
    },
    {
      "id": "tq-20260906-benchfill-muse12-gdpv2-aa",
      "model_slug": "muse-spark-1-2",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA v2",
      "benchmark_version": null,
      "score_numeric": 1631,
      "score_display": "1631 Elo",
      "score_unit": "Elo",
      "tools": true,
      "reasoning_effort": "xhigh",
      "harness": "Stirrup",
      "evaluator": "Artificial Analysis",
      "evaluation_date": "2026-08-05T00:00:00.000Z",
      "source": "https://research.meta.ai/blog/introducing-muse-code-and-muse-spark-1-2",
      "notes": "Provider benchmark result reported in Meta methodology; retained separately from Google's cross-vendor comparison row."
    },
    {
      "id": "meta-20260902-muse12-jobbench",
      "model_slug": "muse-spark-1-2",
      "category": "professional",
      "benchmark_name": "JobBench",
      "benchmark_version": null,
      "score_numeric": 61.6,
      "score_display": "61.6%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "xhigh",
      "harness": "Muse Spark 1.3 launch scorecard",
      "evaluator": "Meta",
      "evaluation_date": "2026-09-02T00:00:00.000Z",
      "source": "https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology",
      "notes": "Muse Spark 1.2 comparison value in Meta's 1.3 scorecard; 1.2 is evaluated at xhigh."
    },
    {
      "id": "tq-20260906-benchfill-muse12-mcp-atlas",
      "model_slug": "muse-spark-1-2",
      "category": "agentic_computer_use",
      "benchmark_name": "MCP Atlas",
      "benchmark_version": null,
      "score_numeric": 90.3,
      "score_display": "90.3%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "xhigh",
      "harness": "Scale AI MCP Atlas",
      "evaluator": "Scale AI",
      "evaluation_date": "2026-08-05T00:00:00.000Z",
      "source": "https://research.meta.ai/blog/introducing-muse-code-and-muse-spark-1-2",
      "notes": null
    },
    {
      "id": "meta-20260902-muse12-mrcr-256-512",
      "model_slug": "muse-spark-1-2",
      "category": "math_reasoning",
      "benchmark_name": "MRCR v2 256K–512K",
      "benchmark_version": "8-needle",
      "score_numeric": 66.3,
      "score_display": "66.3%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "xhigh",
      "harness": "MRCR v2 long-context retrieval",
      "evaluator": "Meta",
      "evaluation_date": "2026-09-02T00:00:00.000Z",
      "source": "https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology",
      "notes": "Long-context retrieval baseline for 1.2."
    },
    {
      "id": "meta-20260902-muse12-mrcr-512-1m",
      "model_slug": "muse-spark-1-2",
      "category": "math_reasoning",
      "benchmark_name": "MRCR v2 512K–1M",
      "benchmark_version": "8-needle",
      "score_numeric": 55.5,
      "score_display": "55.5%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "xhigh",
      "harness": "MRCR v2 long-context retrieval",
      "evaluator": "Meta",
      "evaluation_date": "2026-09-02T00:00:00.000Z",
      "source": "https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology",
      "notes": "Long-context retrieval baseline for 1.2."
    },
    {
      "id": "meta-20260902-muse12-osworld20",
      "model_slug": "muse-spark-1-2",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld 2.0",
      "benchmark_version": "06.24",
      "score_numeric": 47.6,
      "score_display": "47.6%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "xhigh",
      "harness": "Muse Spark 1.3 launch scorecard",
      "evaluator": "Meta",
      "evaluation_date": "2026-09-02T00:00:00.000Z",
      "source": "https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology",
      "notes": "Meta notes the 1.2 OSWorld 2.0 run uses an older benchmark revision than the 1.3 comparison run."
    },
    {
      "id": "meta-20260902-muse12-swe-atlas",
      "model_slug": "muse-spark-1-2",
      "category": "coding",
      "benchmark_name": "SWE-Atlas Codebase QnA",
      "benchmark_version": null,
      "score_numeric": 46.2,
      "score_display": "46.2%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "xhigh",
      "harness": "SWE-Atlas Codebase QnA",
      "evaluator": "Meta",
      "evaluation_date": "2026-09-02T00:00:00.000Z",
      "source": "https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology",
      "notes": "1.2 baseline published in Meta's 1.3 evaluation table."
    },
    {
      "id": "gdm-20260813-muse12-terminal21",
      "model_slug": "muse-spark-1-2",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": null,
      "score_numeric": 82.9,
      "score_display": "82.9%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Gemini 3.7 Flash model-card comparison",
      "evaluator": "Google DeepMind",
      "evaluation_date": "2026-08-13T00:00:00.000Z",
      "source": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
      "notes": "Cross-vendor score reported by Google DeepMind; preserve evaluator and harness."
    },
    {
      "id": "meta-20260902-muse13-agentic-if",
      "model_slug": "muse-spark-1-3",
      "category": "agentic_computer_use",
      "benchmark_name": "Agentic IF Index",
      "benchmark_version": "Meta internal",
      "score_numeric": 57.8,
      "score_display": "57.8%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "Meta internal agentic instruction-following evaluation",
      "evaluator": "Meta",
      "evaluation_date": "2026-09-02T00:00:00.000Z",
      "source": "https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology",
      "notes": "Internal benchmark; useful for same-vendor generation comparison but not an independent leaderboard."
    },
    {
      "id": "tq-20260906-benchfill-muse13-automation",
      "model_slug": "muse-spark-1-3",
      "category": "agentic_computer_use",
      "benchmark_name": "AutomationBench",
      "benchmark_version": null,
      "score_numeric": 49.4,
      "score_display": "49.4%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "Muse Spark 1.3 launch evaluation",
      "evaluator": "Meta",
      "evaluation_date": "2026-09-02T00:00:00.000Z",
      "source": "https://research.meta.ai/blog/introducing-muse-spark-1-3",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-muse13-deepswe",
      "model_slug": "muse-spark-1-3",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": null,
      "score_numeric": 75.4,
      "score_display": "75.4%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "Muse Spark 1.3 launch evaluation",
      "evaluator": "Meta",
      "evaluation_date": "2026-09-02T00:00:00.000Z",
      "source": "https://research.meta.ai/blog/introducing-muse-spark-1-3",
      "notes": null
    },
    {
      "id": "meta-20260902-muse13-deepsearchqa",
      "model_slug": "muse-spark-1-3",
      "category": "agentic_computer_use",
      "benchmark_name": "DeepSearchQA",
      "benchmark_version": null,
      "score_numeric": 89.4,
      "score_display": "89.4%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "Muse Spark 1.3 launch scorecard",
      "evaluator": "Meta",
      "evaluation_date": "2026-09-02T00:00:00.000Z",
      "source": "https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-muse13-gdpv2",
      "model_slug": "muse-spark-1-3",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA v2",
      "benchmark_version": null,
      "score_numeric": 1754,
      "score_display": "1754 Elo",
      "score_unit": "Elo",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "Stirrup",
      "evaluator": "Artificial Analysis",
      "evaluation_date": "2026-09-02T00:00:00.000Z",
      "source": "https://research.meta.ai/blog/introducing-muse-spark-1-3",
      "notes": null
    },
    {
      "id": "meta-20260902-muse13-jobbench",
      "model_slug": "muse-spark-1-3",
      "category": "professional",
      "benchmark_name": "JobBench",
      "benchmark_version": null,
      "score_numeric": 64.9,
      "score_display": "64.9%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "Muse Spark 1.3 launch scorecard",
      "evaluator": "Meta",
      "evaluation_date": "2026-09-02T00:00:00.000Z",
      "source": "https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology",
      "notes": "Professional tool-use evaluation from Meta's launch scorecard."
    },
    {
      "id": "meta-20260902-muse13-mrcr-256-512",
      "model_slug": "muse-spark-1-3",
      "category": "math_reasoning",
      "benchmark_name": "MRCR v2 256K–512K",
      "benchmark_version": "8-needle",
      "score_numeric": 98.5,
      "score_display": "98.5%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "max",
      "harness": "MRCR v2 long-context retrieval",
      "evaluator": "Meta",
      "evaluation_date": "2026-09-02T00:00:00.000Z",
      "source": "https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology",
      "notes": null
    },
    {
      "id": "meta-20260902-muse13-mrcr-512-1m",
      "model_slug": "muse-spark-1-3",
      "category": "math_reasoning",
      "benchmark_name": "MRCR v2 512K–1M",
      "benchmark_version": "8-needle",
      "score_numeric": 98.1,
      "score_display": "98.1%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "max",
      "harness": "MRCR v2 long-context retrieval",
      "evaluator": "Meta",
      "evaluation_date": "2026-09-02T00:00:00.000Z",
      "source": "https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-muse13-osworld20",
      "model_slug": "muse-spark-1-3",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld 2.0",
      "benchmark_version": null,
      "score_numeric": 66.9,
      "score_display": "66.9%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "Muse Spark 1.3 launch evaluation",
      "evaluator": "Meta",
      "evaluation_date": "2026-09-02T00:00:00.000Z",
      "source": "https://research.meta.ai/blog/introducing-muse-spark-1-3",
      "notes": null
    },
    {
      "id": "meta-20260902-muse13-swe-atlas",
      "model_slug": "muse-spark-1-3",
      "category": "coding",
      "benchmark_name": "SWE-Atlas Codebase QnA",
      "benchmark_version": null,
      "score_numeric": 59.4,
      "score_display": "59.4%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "SWE-Atlas Codebase QnA",
      "evaluator": "Meta",
      "evaluation_date": "2026-09-02T00:00:00.000Z",
      "source": "https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology",
      "notes": null
    },
    {
      "id": "tq-20260906-benchfill-muse13-terminal21",
      "model_slug": "muse-spark-1-3",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": null,
      "score_numeric": 88.8,
      "score_display": "88.8%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": "max",
      "harness": "Muse Spark 1.3 launch evaluation",
      "evaluator": "Meta",
      "evaluation_date": "2026-09-02T00:00:00.000Z",
      "source": "https://research.meta.ai/blog/introducing-muse-spark-1-3",
      "notes": null
    },
    {
      "id": "tq-20260907-032-31e6b9f5e66f47e398c5",
      "model_slug": "qwen3-5-0-8b",
      "category": "knowledge",
      "benchmark_name": "MMLU-Pro",
      "benchmark_version": null,
      "score_numeric": 29.7,
      "score_display": "29.7%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "non-thinking",
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-03-02T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-2B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-b2d87626ff6e62885c2c",
      "model_slug": "qwen3-5-0-8b",
      "category": "knowledge",
      "benchmark_name": "MMLU-Pro",
      "benchmark_version": null,
      "score_numeric": 42.3,
      "score_display": "42.3%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-03-02T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-2B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-d8cf4c2809a40fe00b7f",
      "model_slug": "qwen3-5-122b-a10b",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": "context-folding",
      "score_numeric": 63.8,
      "score_display": "63.8%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-24T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-122B-A10B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-52b6c2117be5154e4498",
      "model_slug": "qwen3-5-122b-a10b",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 86.6,
      "score_display": "86.6%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-24T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-122B-A10B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-a1c09a8af4bf585cbf50",
      "model_slug": "qwen3-5-122b-a10b",
      "category": "math_reasoning",
      "benchmark_name": "HMMT",
      "benchmark_version": "Feb. 2025",
      "score_numeric": 91.4,
      "score_display": "91.4%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-24T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-122B-A10B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-cf278ca5ba330abebcdd",
      "model_slug": "qwen3-5-122b-a10b",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 47.5,
      "score_display": "47.5%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-24T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-122B-A10B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-d3fd29c6d98433a0d33d",
      "model_slug": "qwen3-5-122b-a10b",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 25.3,
      "score_display": "25.3%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-24T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-122B-A10B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-9ad03749fdeade6c7bb3",
      "model_slug": "qwen3-5-122b-a10b",
      "category": "coding",
      "benchmark_name": "LiveCodeBench",
      "benchmark_version": "v6",
      "score_numeric": 78.9,
      "score_display": "78.9%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-24T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-122B-A10B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-22459e6713de9a2f0fec",
      "model_slug": "qwen3-5-122b-a10b",
      "category": "knowledge",
      "benchmark_name": "MMLU-Pro",
      "benchmark_version": null,
      "score_numeric": 86.7,
      "score_display": "86.7%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-24T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-122B-A10B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-0e70a8891cda4ba27da8",
      "model_slug": "qwen3-5-122b-a10b",
      "category": "coding",
      "benchmark_name": "SWE-bench Verified",
      "benchmark_version": null,
      "score_numeric": 72,
      "score_display": "72.0%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Qwen evaluation harness",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-24T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-122B-A10B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-2b0e0d19dc15a0b3829d",
      "model_slug": "qwen3-5-122b-a10b",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.0",
      "benchmark_version": null,
      "score_numeric": 49.4,
      "score_display": "49.4%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Qwen evaluation harness",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-24T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-122B-A10B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-8daa7b884104d546aefe",
      "model_slug": "qwen3-5-27b",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": "context-folding",
      "score_numeric": 61,
      "score_display": "61.0%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-24T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-122B-A10B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-9208d480d01377a7a735",
      "model_slug": "qwen3-5-27b",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 85.5,
      "score_display": "85.5%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-24T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-122B-A10B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-91d6e4b00c8b9ff657d1",
      "model_slug": "qwen3-5-27b",
      "category": "math_reasoning",
      "benchmark_name": "HMMT",
      "benchmark_version": "Feb. 2025",
      "score_numeric": 92,
      "score_display": "92.0%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-24T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-122B-A10B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-47af522b914a7558d6de",
      "model_slug": "qwen3-5-27b",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 48.5,
      "score_display": "48.5%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-24T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-122B-A10B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-56a510e36b557b42b786",
      "model_slug": "qwen3-5-27b",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 24.3,
      "score_display": "24.3%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-24T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-122B-A10B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-0e5934e076c889115853",
      "model_slug": "qwen3-5-27b",
      "category": "coding",
      "benchmark_name": "LiveCodeBench",
      "benchmark_version": "v6",
      "score_numeric": 80.7,
      "score_display": "80.7%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-24T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-122B-A10B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-dc2c3e26b8d6b94d6203",
      "model_slug": "qwen3-5-27b",
      "category": "knowledge",
      "benchmark_name": "MMLU-Pro",
      "benchmark_version": null,
      "score_numeric": 86.1,
      "score_display": "86.1%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-24T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-122B-A10B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-eea73e97176b445c0e1a",
      "model_slug": "qwen3-5-27b",
      "category": "coding",
      "benchmark_name": "SWE-bench Verified",
      "benchmark_version": null,
      "score_numeric": 72.4,
      "score_display": "72.4%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Qwen evaluation harness",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-24T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-122B-A10B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-71f568fb5ba23f89ef55",
      "model_slug": "qwen3-5-27b",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.0",
      "benchmark_version": null,
      "score_numeric": 41.6,
      "score_display": "41.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Qwen evaluation harness",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-24T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-122B-A10B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-52c3333c6985951d94ee",
      "model_slug": "qwen3-5-2b",
      "category": "math_reasoning",
      "benchmark_name": "HMMT",
      "benchmark_version": "Feb. 2025",
      "score_numeric": 22.9,
      "score_display": "22.9%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-03-02T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-2B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-c723c78cd2d169e00148",
      "model_slug": "qwen3-5-2b",
      "category": "math_reasoning",
      "benchmark_name": "HMMT",
      "benchmark_version": "Nov. 2025",
      "score_numeric": 19.6,
      "score_display": "19.6%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-03-02T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-2B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-77c0178c1c99eec54df9",
      "model_slug": "qwen3-5-2b",
      "category": "knowledge",
      "benchmark_name": "MMLU-Pro",
      "benchmark_version": null,
      "score_numeric": 66.5,
      "score_display": "66.5%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "thinking",
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-03-02T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-2B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-e39e0e7ef48282059c32",
      "model_slug": "qwen3-5-2b",
      "category": "knowledge",
      "benchmark_name": "MMLU-Pro",
      "benchmark_version": null,
      "score_numeric": 55.3,
      "score_display": "55.3%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": "non-thinking",
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-03-02T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-2B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-5f249573bf754ef91d7b",
      "model_slug": "qwen3-5-35b-a3b",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": "context-folding",
      "score_numeric": 61,
      "score_display": "61.0%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-24T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-122B-A10B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-1db72357823df50184b6",
      "model_slug": "qwen3-5-35b-a3b",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 84.2,
      "score_display": "84.2%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-24T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-122B-A10B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-5f86c859f0400dd1432d",
      "model_slug": "qwen3-5-35b-a3b",
      "category": "math_reasoning",
      "benchmark_name": "HMMT",
      "benchmark_version": "Feb. 2025",
      "score_numeric": 89,
      "score_display": "89.0%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-24T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-122B-A10B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-947def9dba3c36f332e0",
      "model_slug": "qwen3-5-35b-a3b",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 22.4,
      "score_display": "22.4%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-24T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-122B-A10B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-a5b0e43981a03c2c5a39",
      "model_slug": "qwen3-5-35b-a3b",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 47.4,
      "score_display": "47.4%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-24T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-122B-A10B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-4130f7a35598599e2b86",
      "model_slug": "qwen3-5-35b-a3b",
      "category": "coding",
      "benchmark_name": "LiveCodeBench",
      "benchmark_version": "v6",
      "score_numeric": 74.6,
      "score_display": "74.6%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-24T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-122B-A10B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-5fd375e21f9d116cf399",
      "model_slug": "qwen3-5-35b-a3b",
      "category": "knowledge",
      "benchmark_name": "MMLU-Pro",
      "benchmark_version": null,
      "score_numeric": 85.3,
      "score_display": "85.3%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-24T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-122B-A10B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-d2747b2b3dae799e41e1",
      "model_slug": "qwen3-5-35b-a3b",
      "category": "coding",
      "benchmark_name": "SWE-bench Verified",
      "benchmark_version": null,
      "score_numeric": 69.2,
      "score_display": "69.2%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Qwen evaluation harness",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-24T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-122B-A10B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-f5725e85fb63efe95220",
      "model_slug": "qwen3-5-35b-a3b",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.0",
      "benchmark_version": null,
      "score_numeric": 40.5,
      "score_display": "40.5%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Qwen evaluation harness",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-24T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-122B-A10B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-f591d20447db2ff142f7",
      "model_slug": "qwen3-5-397b-a17b",
      "category": "math_reasoning",
      "benchmark_name": "AIME",
      "benchmark_version": "2026",
      "score_numeric": 91.3,
      "score_display": "91.3%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-15T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.5",
      "notes": null
    },
    {
      "id": "tq-20260907-032-6784e3949ad69e9d4d86",
      "model_slug": "qwen3-5-397b-a17b",
      "category": "agentic_computer_use",
      "benchmark_name": "BrowseComp",
      "benchmark_version": "context-folding",
      "score_numeric": 69,
      "score_display": "69.0%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-15T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.5",
      "notes": null
    },
    {
      "id": "tq-20260907-032-01032834aefa849a2ae8",
      "model_slug": "qwen3-5-397b-a17b",
      "category": "knowledge",
      "benchmark_name": "HLE-Verified",
      "benchmark_version": null,
      "score_numeric": 37.6,
      "score_display": "37.6%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-15T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.5",
      "notes": null
    },
    {
      "id": "tq-20260907-032-90abc219a6dca1ed4c49",
      "model_slug": "qwen3-5-397b-a17b",
      "category": "math_reasoning",
      "benchmark_name": "HMMT",
      "benchmark_version": "Feb. 2025",
      "score_numeric": 94.8,
      "score_display": "94.8%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-15T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.5",
      "notes": null
    },
    {
      "id": "tq-20260907-032-dad3b46d72c91395da68",
      "model_slug": "qwen3-5-397b-a17b",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 28.7,
      "score_display": "28.7%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-15T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.5",
      "notes": null
    },
    {
      "id": "tq-20260907-032-48ec491da47629400e35",
      "model_slug": "qwen3-5-397b-a17b",
      "category": "coding",
      "benchmark_name": "LiveCodeBench",
      "benchmark_version": "v6",
      "score_numeric": 83.6,
      "score_display": "83.6%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-15T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.5",
      "notes": null
    },
    {
      "id": "tq-20260907-032-129a67393c2634280b3b",
      "model_slug": "qwen3-5-397b-a17b",
      "category": "knowledge",
      "benchmark_name": "MMLU-Pro",
      "benchmark_version": null,
      "score_numeric": 87.8,
      "score_display": "87.8%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-15T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.5",
      "notes": null
    },
    {
      "id": "tq-20260907-032-d0a8c9347928a53b2a6a",
      "model_slug": "qwen3-5-397b-a17b",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": "Public",
      "score_numeric": 50.9,
      "score_display": "50.9%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Qwen agent scaffold",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-15T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.6-27b",
      "notes": null
    },
    {
      "id": "tq-20260907-032-ba2e634e86be0b1efc0f",
      "model_slug": "qwen3-5-397b-a17b",
      "category": "coding",
      "benchmark_name": "SWE-bench Verified",
      "benchmark_version": null,
      "score_numeric": 76.2,
      "score_display": "76.2%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Qwen agent scaffold",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-15T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.6-27b",
      "notes": null
    },
    {
      "id": "tq-20260907-032-9cbdc7fef7eb546d6aea",
      "model_slug": "qwen3-5-397b-a17b",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.0",
      "benchmark_version": null,
      "score_numeric": 52.5,
      "score_display": "52.5%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Harbor / Terminus-2",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-02-15T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.6-27b",
      "notes": null
    },
    {
      "id": "tq-20260907-032-f9ae0574f244ead1c4c1",
      "model_slug": "qwen3-5-4b",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 76.2,
      "score_display": "76.2%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-03-02T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-9B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-10eeeeeadf4a7837bfbf",
      "model_slug": "qwen3-5-4b",
      "category": "math_reasoning",
      "benchmark_name": "HMMT",
      "benchmark_version": "Feb. 2025",
      "score_numeric": 74,
      "score_display": "74.0%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-03-02T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-9B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-55b7abfbd674a1a54557",
      "model_slug": "qwen3-5-4b",
      "category": "coding",
      "benchmark_name": "LiveCodeBench",
      "benchmark_version": "v6",
      "score_numeric": 55.8,
      "score_display": "55.8%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-03-02T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-9B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-b0d5b5ff8bb40d913738",
      "model_slug": "qwen3-5-4b",
      "category": "knowledge",
      "benchmark_name": "MMLU-Pro",
      "benchmark_version": null,
      "score_numeric": 79.1,
      "score_display": "79.1%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-03-02T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-9B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-1ef9fbbde9d7760d57bc",
      "model_slug": "qwen3-5-4b",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld-Verified",
      "benchmark_version": null,
      "score_numeric": 35.6,
      "score_display": "35.6%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-03-02T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-9B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-2dbe8cbed7fd2ad75683",
      "model_slug": "qwen3-5-9b",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 81.7,
      "score_display": "81.7%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-03-02T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-9B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-165055fefe219fd763d7",
      "model_slug": "qwen3-5-9b",
      "category": "math_reasoning",
      "benchmark_name": "HMMT",
      "benchmark_version": "Feb. 2025",
      "score_numeric": 83.2,
      "score_display": "83.2%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-03-02T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-9B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-620a9636dd98a4b951f6",
      "model_slug": "qwen3-5-9b",
      "category": "coding",
      "benchmark_name": "LiveCodeBench",
      "benchmark_version": "v6",
      "score_numeric": 65.6,
      "score_display": "65.6%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-03-02T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-9B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-a47c549ff162e7b568dc",
      "model_slug": "qwen3-5-9b",
      "category": "knowledge",
      "benchmark_name": "MMLU-Pro",
      "benchmark_version": null,
      "score_numeric": 82.5,
      "score_display": "82.5%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-03-02T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-9B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-18e730e89248ff764f37",
      "model_slug": "qwen3-5-9b",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld-Verified",
      "benchmark_version": null,
      "score_numeric": 41.8,
      "score_display": "41.8%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-03-02T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.5-9B",
      "notes": null
    },
    {
      "id": "tq-20260907-029-qwen3-6-27b-aime",
      "model_slug": "qwen3-6-27b",
      "category": "math_reasoning",
      "benchmark_name": "AIME",
      "benchmark_version": "2026",
      "score_numeric": 94.1,
      "score_display": "94.1%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-04-21T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.6-27b",
      "notes": null
    },
    {
      "id": "qwen3-6-27b-gpqa-diamond-87-8-none-none-none-qwen",
      "model_slug": "qwen3-6-27b",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 87.8,
      "score_display": "87.8",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Qwen",
      "evaluation_date": null,
      "source": "https://huggingface.co/Qwen/Qwen3.6-27B",
      "notes": null
    },
    {
      "id": "tq-20260907-029-qwen3-6-27b-gpqa",
      "model_slug": "qwen3-6-27b",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 87.8,
      "score_display": "87.8%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-04-21T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.6-27b",
      "notes": null
    },
    {
      "id": "tq-20260907-029-qwen3-6-27b-hmmt",
      "model_slug": "qwen3-6-27b",
      "category": "math_reasoning",
      "benchmark_name": "HMMT",
      "benchmark_version": "Feb. 2026",
      "score_numeric": 84.3,
      "score_display": "84.3%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-04-21T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.6-27b",
      "notes": null
    },
    {
      "id": "tq-20260907-029-qwen3-6-27b-hle",
      "model_slug": "qwen3-6-27b",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 24,
      "score_display": "24.0%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-04-21T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.6-27b",
      "notes": null
    },
    {
      "id": "qwen3-6-27b-livecodebench-v6-83-9-none-none-none-qwen",
      "model_slug": "qwen3-6-27b",
      "category": "coding",
      "benchmark_name": "LiveCodeBench",
      "benchmark_version": "v6",
      "score_numeric": 83.9,
      "score_display": "83.9",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Qwen",
      "evaluation_date": null,
      "source": "https://huggingface.co/Qwen/Qwen3.6-27B",
      "notes": null
    },
    {
      "id": "tq-20260907-029-qwen3-6-27b-lcb",
      "model_slug": "qwen3-6-27b",
      "category": "coding",
      "benchmark_name": "LiveCodeBench",
      "benchmark_version": "v6",
      "score_numeric": 83.9,
      "score_display": "83.9%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-04-21T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.6-27b",
      "notes": null
    },
    {
      "id": "qwen3-6-27b-mmlu-pro-86-2-none-none-none-qwen",
      "model_slug": "qwen3-6-27b",
      "category": "knowledge",
      "benchmark_name": "MMLU-Pro",
      "benchmark_version": null,
      "score_numeric": 86.2,
      "score_display": "86.2",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Qwen",
      "evaluation_date": null,
      "source": "https://huggingface.co/Qwen/Qwen3.6-27B",
      "notes": null
    },
    {
      "id": "qwen3-6-27b-swe-bench-pro-53-5-none-none-none-qwen",
      "model_slug": "qwen3-6-27b",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": null,
      "score_numeric": 53.5,
      "score_display": "53.5",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Qwen",
      "evaluation_date": null,
      "source": "https://huggingface.co/Qwen/Qwen3.6-27B",
      "notes": null
    },
    {
      "id": "tq-20260907-029-qwen3-6-27b-swepro",
      "model_slug": "qwen3-6-27b",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": "Public",
      "score_numeric": 53.5,
      "score_display": "53.5%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Qwen agent scaffold",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-04-21T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.6-27b",
      "notes": null
    },
    {
      "id": "qwen3-6-27b-swe-bench-verified-77-2-none-none-none-qwen",
      "model_slug": "qwen3-6-27b",
      "category": "coding",
      "benchmark_name": "SWE-bench Verified",
      "benchmark_version": null,
      "score_numeric": 77.2,
      "score_display": "77.2",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Qwen",
      "evaluation_date": null,
      "source": "https://huggingface.co/Qwen/Qwen3.6-27B",
      "notes": null
    },
    {
      "id": "tq-20260907-029-qwen3-6-27b-swev",
      "model_slug": "qwen3-6-27b",
      "category": "coding",
      "benchmark_name": "SWE-bench Verified",
      "benchmark_version": null,
      "score_numeric": 77.2,
      "score_display": "77.2%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Qwen agent scaffold",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-04-21T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.6-27b",
      "notes": null
    },
    {
      "id": "qwen3-6-27b-skillsbench-v1-48-2-none-none-opencode-qwen",
      "model_slug": "qwen3-6-27b",
      "category": "agentic_computer_use",
      "benchmark_name": "SkillsBench",
      "benchmark_version": "v1",
      "score_numeric": 48.2,
      "score_display": "48.2",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": "OpenCode",
      "evaluator": "Qwen",
      "evaluation_date": null,
      "source": "https://huggingface.co/Qwen/Qwen3.6-27B",
      "notes": null
    },
    {
      "id": "qwen3-6-27b-terminal-bench-2-0-59-3-none-none-none-qwen",
      "model_slug": "qwen3-6-27b",
      "category": "coding",
      "benchmark_name": "Terminal-Bench",
      "benchmark_version": "2.0",
      "score_numeric": 59.3,
      "score_display": "59.3",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Qwen",
      "evaluation_date": null,
      "source": "https://huggingface.co/Qwen/Qwen3.6-27B",
      "notes": "Model-card result marked with footnote; preserve harness notes from source."
    },
    {
      "id": "tq-20260907-029-qwen3-6-27b-term20",
      "model_slug": "qwen3-6-27b",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.0",
      "benchmark_version": null,
      "score_numeric": 59.3,
      "score_display": "59.3%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Harbor / Terminus-2",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-04-21T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.6-27b",
      "notes": null
    },
    {
      "id": "tq-20260907-029-qwen3-6-35b-a3b-aime",
      "model_slug": "qwen3-6-35b-a3b",
      "category": "math_reasoning",
      "benchmark_name": "AIME",
      "benchmark_version": "2026",
      "score_numeric": 92.7,
      "score_display": "92.7%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-04-15T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.6-35B-A3B",
      "notes": null
    },
    {
      "id": "tq-20260907-029-qwen3-6-35b-a3b-gpqa",
      "model_slug": "qwen3-6-35b-a3b",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 86,
      "score_display": "86.0%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-04-15T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.6-35B-A3B",
      "notes": null
    },
    {
      "id": "tq-20260907-029-qwen3-6-35b-a3b-hmmt",
      "model_slug": "qwen3-6-35b-a3b",
      "category": "math_reasoning",
      "benchmark_name": "HMMT",
      "benchmark_version": "Feb. 2026",
      "score_numeric": 83.6,
      "score_display": "83.6%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-04-15T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.6-35B-A3B",
      "notes": null
    },
    {
      "id": "tq-20260907-029-qwen3-6-35b-a3b-hle",
      "model_slug": "qwen3-6-35b-a3b",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 21.4,
      "score_display": "21.4%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-04-15T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.6-35B-A3B",
      "notes": null
    },
    {
      "id": "tq-20260907-029-qwen3-6-35b-a3b-lcb",
      "model_slug": "qwen3-6-35b-a3b",
      "category": "coding",
      "benchmark_name": "LiveCodeBench",
      "benchmark_version": "v6",
      "score_numeric": 80.4,
      "score_display": "80.4%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-04-15T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.6-35B-A3B",
      "notes": null
    },
    {
      "id": "tq-20260907-029-qwen3-6-35b-a3b-mcp",
      "model_slug": "qwen3-6-35b-a3b",
      "category": "agentic_computer_use",
      "benchmark_name": "MCP Atlas",
      "benchmark_version": "Public Set",
      "score_numeric": 62.8,
      "score_display": "62.8%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-04-15T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.6-35B-A3B",
      "notes": null
    },
    {
      "id": "tq-20260907-029-qwen3-6-35b-a3b-swepro",
      "model_slug": "qwen3-6-35b-a3b",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": "Public",
      "score_numeric": 49.5,
      "score_display": "49.5%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Qwen agent scaffold",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-04-15T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.6-35B-A3B",
      "notes": null
    },
    {
      "id": "tq-20260907-029-qwen3-6-35b-a3b-swev",
      "model_slug": "qwen3-6-35b-a3b",
      "category": "coding",
      "benchmark_name": "SWE-bench Verified",
      "benchmark_version": null,
      "score_numeric": 73.4,
      "score_display": "73.4%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Qwen agent scaffold",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-04-15T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.6-35B-A3B",
      "notes": null
    },
    {
      "id": "tq-20260907-029-qwen3-6-35b-a3b-term20",
      "model_slug": "qwen3-6-35b-a3b",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.0",
      "benchmark_version": null,
      "score_numeric": 51.5,
      "score_display": "51.5%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Harbor / Terminus-2",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-04-15T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.6-35B-A3B",
      "notes": null
    },
    {
      "id": "tq-20260907-032-e8b8033e002f3048ed59",
      "model_slug": "qwen3-7-plus",
      "category": "agentic_computer_use",
      "benchmark_name": "Agents' Last Exam",
      "benchmark_version": "Pass@1",
      "score_numeric": 13.2,
      "score_display": "13.2%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-26T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.8-flash-next",
      "notes": "Cross-generation rerun published by Qwen in the Qwen3.8-Flash-Next launch evaluation."
    },
    {
      "id": "tq-20260907-032-f889024d1c99427c0dda",
      "model_slug": "qwen3-7-plus",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": null,
      "score_numeric": 16.5,
      "score_display": "16.5%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Claude Code / mini-SWE-agent",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-26T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.8-flash-next",
      "notes": "Cross-generation rerun published by Qwen in the Qwen3.8-Flash-Next launch evaluation."
    },
    {
      "id": "tq-20260907-032-2e09bb5526855bdd154c",
      "model_slug": "qwen3-7-plus",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 90.3,
      "score_display": "90.3%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-26T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.8-flash-next",
      "notes": "Cross-generation rerun published by Qwen in the Qwen3.8-Flash-Next launch evaluation."
    },
    {
      "id": "tq-20260907-032-022483c866d01ebb77c9",
      "model_slug": "qwen3-7-plus",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 34.7,
      "score_display": "34.7%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-26T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.8-flash-next",
      "notes": "Cross-generation rerun published by Qwen in the Qwen3.8-Flash-Next launch evaluation."
    },
    {
      "id": "tq-20260907-032-8b53e70131687230011d",
      "model_slug": "qwen3-7-plus",
      "category": "coding",
      "benchmark_name": "LiveCodeBench",
      "benchmark_version": "v6",
      "score_numeric": 89.6,
      "score_display": "89.6%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-26T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.8-flash-next",
      "notes": "Cross-generation rerun published by Qwen in the Qwen3.8-Flash-Next launch evaluation."
    },
    {
      "id": "tq-20260907-032-3ca582746dcc464a9d60",
      "model_slug": "qwen3-7-plus",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld 2.0",
      "benchmark_version": "Binary",
      "score_numeric": 2.8,
      "score_display": "2.8%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-26T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.8-flash-next",
      "notes": "Cross-generation rerun published by Qwen in the Qwen3.8-Flash-Next launch evaluation."
    },
    {
      "id": "tq-20260907-032-52e76105f7c5084ef20c",
      "model_slug": "qwen3-7-plus",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld 2.0",
      "benchmark_version": "Partial",
      "score_numeric": 21.5,
      "score_display": "21.5%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-26T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.8-flash-next",
      "notes": "Cross-generation rerun published by Qwen in the Qwen3.8-Flash-Next launch evaluation."
    },
    {
      "id": "tq-20260907-032-bedf4fcd4c99cc7741ff",
      "model_slug": "qwen3-7-plus",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": "Public",
      "score_numeric": 55.8,
      "score_display": "55.8%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Claude Code",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-26T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.8-flash-next",
      "notes": "Cross-generation rerun published by Qwen in the Qwen3.8-Flash-Next launch evaluation."
    },
    {
      "id": "tq-20260907-032-d86311612d1609d2c70d",
      "model_slug": "qwen3-7-plus",
      "category": "agentic_computer_use",
      "benchmark_name": "Toolathlon",
      "benchmark_version": "Verified",
      "score_numeric": 50.6,
      "score_display": "50.6%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-26T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.8-flash-next",
      "notes": "Cross-generation rerun published by Qwen in the Qwen3.8-Flash-Next launch evaluation."
    },
    {
      "id": "tq-20260907-029-qwen3-8-2-4t-a95b-deepswe",
      "model_slug": "qwen3-8-2-4t-a95b",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": null,
      "score_numeric": 56.6,
      "score_display": "56.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-12T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B",
      "notes": "Checkpoint score published by Qwen."
    },
    {
      "id": "tq-20260907-029-qwen3-8-2-4t-a95b-gpqa",
      "model_slug": "qwen3-8-2-4t-a95b",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 92.6,
      "score_display": "92.6%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-12T00:00:00.000Z",
      "source": "https://www.alibabacloud.com/help/en/model-studio/qwen3-8-2-4t-a95b",
      "notes": null
    },
    {
      "id": "tq-20260907-029-qwen3-8-2-4t-a95b-osworld",
      "model_slug": "qwen3-8-2-4t-a95b",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld",
      "benchmark_version": null,
      "score_numeric": 86.1,
      "score_display": "86.1%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-12T00:00:00.000Z",
      "source": "https://www.alibabacloud.com/help/en/model-studio/qwen3-8-2-4t-a95b",
      "notes": null
    },
    {
      "id": "qwen3-8-2-4t-a95b-swe-bench-pro-67-7-none-none-none-qwen",
      "model_slug": "qwen3-8-2-4t-a95b",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": null,
      "score_numeric": 67.7,
      "score_display": "67.7",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Qwen",
      "evaluation_date": null,
      "source": "https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B",
      "notes": null
    },
    {
      "id": "qwen3-8-2-4t-a95b-terminal-bench-2-1-86-6-none-none-none-qwen",
      "model_slug": "qwen3-8-2-4t-a95b",
      "category": "coding",
      "benchmark_name": "Terminal-Bench",
      "benchmark_version": "2.1",
      "score_numeric": 86.6,
      "score_display": "86.6",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Qwen",
      "evaluation_date": null,
      "source": "https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B",
      "notes": null
    },
    {
      "id": "tq-20260907-029-qwen3-8-2-4t-a95b-term21",
      "model_slug": "qwen3-8-2-4t-a95b",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": null,
      "score_numeric": 86.6,
      "score_display": "86.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-12T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B",
      "notes": "Checkpoint score published by Qwen."
    },
    {
      "id": "tq-20260907-qwen3827b-ale",
      "model_slug": "qwen3-8-27b",
      "category": "agentic_computer_use",
      "benchmark_name": "Agents' Last Exam",
      "benchmark_version": "Pass@1",
      "score_numeric": 20.4,
      "score_display": "20.4%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Qwen3.8-27B official evaluation",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-17T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
      "notes": "Official table also reports overall ALE score 42.9; canonical comparison stores pass@1."
    },
    {
      "id": "tq-20260907-qwen3827b-deepswe",
      "model_slug": "qwen3-8-27b",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": null,
      "score_numeric": 42.2,
      "score_display": "42.2%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Claude Code; 256K context",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-17T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
      "notes": null
    },
    {
      "id": "tq-20260907-qwen3827b-gpqa",
      "model_slug": "qwen3-8-27b",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 89.2,
      "score_display": "89.2%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": "Qwen3.8-27B official evaluation",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-17T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
      "notes": null
    },
    {
      "id": "tq-20260907-qwen3827b-hle",
      "model_slug": "qwen3-8-27b",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 30.8,
      "score_display": "30.8%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": "Qwen3.8-27B official evaluation",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-17T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
      "notes": null
    },
    {
      "id": "tq-20260907-qwen3827b-livecode",
      "model_slug": "qwen3-8-27b",
      "category": "coding",
      "benchmark_name": "LiveCodeBench",
      "benchmark_version": "v6",
      "score_numeric": 90.3,
      "score_display": "90.3%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": "Qwen3.8-27B official evaluation",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-17T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
      "notes": null
    },
    {
      "id": "tq-20260907-qwen3827b-osworld20-binary",
      "model_slug": "qwen3-8-27b",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld 2.0",
      "benchmark_version": "Binary",
      "score_numeric": 19.4,
      "score_display": "19.4%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Qwen3.8-27B multimodal agent evaluation",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-17T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
      "notes": null
    },
    {
      "id": "tq-20260907-qwen3827b-osworld20-partial",
      "model_slug": "qwen3-8-27b",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld 2.0",
      "benchmark_version": "Partial",
      "score_numeric": 48,
      "score_display": "48.0%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Qwen3.8-27B multimodal agent evaluation",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-17T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
      "notes": null
    },
    {
      "id": "tq-20260907-qwen3827b-osworld-verified",
      "model_slug": "qwen3-8-27b",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld-Verified",
      "benchmark_version": null,
      "score_numeric": 84.3,
      "score_display": "84.3%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Qwen3.8-27B multimodal agent evaluation",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-17T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
      "notes": null
    },
    {
      "id": "tq-20260907-qwen3827b-swe-pro",
      "model_slug": "qwen3-8-27b",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": null,
      "score_numeric": 61.7,
      "score_display": "61.7%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Claude Code; 256K context",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-17T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
      "notes": null
    },
    {
      "id": "tq-20260907-qwen3827b-terminal21",
      "model_slug": "qwen3-8-27b",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": null,
      "score_numeric": 73,
      "score_display": "73.0%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Terminus",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-17T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
      "notes": null
    },
    {
      "id": "tq-20260907-qwen3827b-toolathlon",
      "model_slug": "qwen3-8-27b",
      "category": "agentic_computer_use",
      "benchmark_name": "Toolathlon",
      "benchmark_version": "Verified",
      "score_numeric": 67.1,
      "score_display": "67.1%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Qwen3.8-27B official evaluation",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-17T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
      "notes": null
    },
    {
      "id": "tq-20260907-029-qwen3-8-flash-next-ale",
      "model_slug": "qwen3-8-flash-next",
      "category": "agentic_computer_use",
      "benchmark_name": "Agents' Last Exam",
      "benchmark_version": "Pass@1",
      "score_numeric": 24.3,
      "score_display": "24.3%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-26T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.8-flash-next",
      "notes": null
    },
    {
      "id": "qwen3-8-flash-next-androidworld-84-5-none-none-none-qwen",
      "model_slug": "qwen3-8-flash-next",
      "category": "agentic_computer_use",
      "benchmark_name": "AndroidWorld",
      "benchmark_version": null,
      "score_numeric": 84.5,
      "score_display": "84.5",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Qwen",
      "evaluation_date": null,
      "source": "https://qwen.ai/blog?id=qwen3.8-flash-next",
      "notes": null
    },
    {
      "id": "qwen3-8-flash-next-deepswe-1-1-58-7-none-none-none-qwen",
      "model_slug": "qwen3-8-flash-next",
      "category": "coding",
      "benchmark_name": "DeepSWE",
      "benchmark_version": "1.1",
      "score_numeric": 58.7,
      "score_display": "58.7",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Qwen",
      "evaluation_date": null,
      "source": "https://qwen.ai/blog?id=qwen3.8-flash-next",
      "notes": null
    },
    {
      "id": "tq-20260907-029-qwen3-8-flash-next-deepswe",
      "model_slug": "qwen3-8-flash-next",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": null,
      "score_numeric": 58.7,
      "score_display": "58.7%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-26T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.8-flash-next",
      "notes": null
    },
    {
      "id": "tq-20260907-029-qwen3-8-flash-next-gpqa",
      "model_slug": "qwen3-8-flash-next",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 91.7,
      "score_display": "91.7%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-26T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.8-flash-next",
      "notes": null
    },
    {
      "id": "tq-20260907-029-qwen3-8-flash-next-hle",
      "model_slug": "qwen3-8-flash-next",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 35.9,
      "score_display": "35.9%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-26T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.8-flash-next",
      "notes": null
    },
    {
      "id": "tq-20260907-029-qwen3-8-flash-next-lcb",
      "model_slug": "qwen3-8-flash-next",
      "category": "coding",
      "benchmark_name": "LiveCodeBench",
      "benchmark_version": "v6",
      "score_numeric": 91.9,
      "score_display": "91.9%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-26T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.8-flash-next",
      "notes": null
    },
    {
      "id": "qwen3-8-flash-next-osworld-2-0-binary-19-4-none-none-none-qwen",
      "model_slug": "qwen3-8-flash-next",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld",
      "benchmark_version": "2.0 binary",
      "score_numeric": 19.4,
      "score_display": "19.4",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Qwen",
      "evaluation_date": null,
      "source": "https://qwen.ai/blog?id=qwen3.8-flash-next",
      "notes": null
    },
    {
      "id": "qwen3-8-flash-next-osworld-2-0-partial-52-3-none-none-none-qwen",
      "model_slug": "qwen3-8-flash-next",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld",
      "benchmark_version": "2.0 partial",
      "score_numeric": 52.3,
      "score_display": "52.3",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Qwen",
      "evaluation_date": null,
      "source": "https://qwen.ai/blog?id=qwen3.8-flash-next",
      "notes": null
    },
    {
      "id": "tq-20260907-029-qwen3-8-flash-next-osbinary",
      "model_slug": "qwen3-8-flash-next",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld 2.0",
      "benchmark_version": "Binary",
      "score_numeric": 19.4,
      "score_display": "19.4%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-26T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.8-flash-next",
      "notes": null
    },
    {
      "id": "tq-20260907-029-qwen3-8-flash-next-ospartial",
      "model_slug": "qwen3-8-flash-next",
      "category": "agentic_computer_use",
      "benchmark_name": "OSWorld 2.0",
      "benchmark_version": "Partial",
      "score_numeric": 52.3,
      "score_display": "52.3%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-26T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.8-flash-next",
      "notes": null
    },
    {
      "id": "qwen3-8-flash-next-swe-bench-pro-62-5-none-none-none-qwen",
      "model_slug": "qwen3-8-flash-next",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": null,
      "score_numeric": 62.5,
      "score_display": "62.5",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Qwen",
      "evaluation_date": null,
      "source": "https://qwen.ai/blog?id=qwen3.8-flash-next",
      "notes": null
    },
    {
      "id": "tq-20260907-029-qwen3-8-flash-next-swepro",
      "model_slug": "qwen3-8-flash-next",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": "Public",
      "score_numeric": 62.5,
      "score_display": "62.5%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Claude Code",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-26T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.8-flash-next",
      "notes": null
    },
    {
      "id": "tq-20260907-029-qwen3-8-flash-next-toolathlon",
      "model_slug": "qwen3-8-flash-next",
      "category": "agentic_computer_use",
      "benchmark_name": "Toolathlon",
      "benchmark_version": "Verified",
      "score_numeric": 73.5,
      "score_display": "73.5%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-26T00:00:00.000Z",
      "source": "https://qwen.ai/blog?id=qwen3.8-flash-next",
      "notes": null
    },
    {
      "id": "qwen3-8-flash-next-toolathlon-verified-73-5-none-none-none-qwen",
      "model_slug": "qwen3-8-flash-next",
      "category": "agentic_computer_use",
      "benchmark_name": "Toolathlon Verified",
      "benchmark_version": null,
      "score_numeric": 73.5,
      "score_display": "73.5",
      "score_unit": "%",
      "tools": null,
      "reasoning_effort": null,
      "harness": null,
      "evaluator": "Qwen",
      "evaluation_date": null,
      "source": "https://qwen.ai/blog?id=qwen3.8-flash-next",
      "notes": null
    },
    {
      "id": "tq-20260907-qwen38max-ale",
      "model_slug": "qwen3-8-max",
      "category": "agentic_computer_use",
      "benchmark_name": "Agents' Last Exam",
      "benchmark_version": "Pass",
      "score_numeric": 27,
      "score_display": "27.0%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Qwen3.8 official evaluation",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-02T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B",
      "notes": "Official table also reports overall ALE score 52.4; canonical comparison stores the pass rate."
    },
    {
      "id": "tq-20260907-qwen38max-automation",
      "model_slug": "qwen3-8-max",
      "category": "agentic_computer_use",
      "benchmark_name": "AutomationBench",
      "benchmark_version": null,
      "score_numeric": 27.3,
      "score_display": "27.3%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Qwen3.8 official evaluation",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-02T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B",
      "notes": null
    },
    {
      "id": "tq-20260907-qwen38max-deepswe",
      "model_slug": "qwen3-8-max",
      "category": "coding",
      "benchmark_name": "DeepSWE v1.1",
      "benchmark_version": null,
      "score_numeric": 56.6,
      "score_display": "56.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Qwen3.8 official evaluation",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-02T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B",
      "notes": null
    },
    {
      "id": "tq-20260907-qwen38max-gdpv2",
      "model_slug": "qwen3-8-max",
      "category": "agentic_computer_use",
      "benchmark_name": "GDPval-AA v2",
      "benchmark_version": null,
      "score_numeric": 1739,
      "score_display": "1739 Elo",
      "score_unit": "Elo",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Artificial Analysis",
      "evaluator": "Artificial Analysis",
      "evaluation_date": "2026-08-28T00:00:00.000Z",
      "source": "https://huggingface.co/zai-org/GLM-5.3",
      "notes": "Cross-vendor value reported in Z.ai's official GLM-5.3 comparison table; kept distinct from Qwen's direct benchmark rows."
    },
    {
      "id": "tq-20260907-qwen38max-gpqa",
      "model_slug": "qwen3-8-max",
      "category": "knowledge",
      "benchmark_name": "GPQA Diamond",
      "benchmark_version": null,
      "score_numeric": 92.6,
      "score_display": "92.6%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": "Qwen3.8 official evaluation",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-02T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B",
      "notes": null
    },
    {
      "id": "tq-20260907-qwen38max-hle-no-tools",
      "model_slug": "qwen3-8-max",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 43.6,
      "score_display": "43.6%",
      "score_unit": "percent",
      "tools": false,
      "reasoning_effort": null,
      "harness": "Qwen3.8 official evaluation",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-02T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B",
      "notes": null
    },
    {
      "id": "tq-20260907-qwen38max-hle-tools",
      "model_slug": "qwen3-8-max",
      "category": "knowledge",
      "benchmark_name": "Humanity's Last Exam",
      "benchmark_version": null,
      "score_numeric": 56.2,
      "score_display": "56.2%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Qwen3.8 official evaluation",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-02T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B",
      "notes": null
    },
    {
      "id": "tq-20260907-qwen38max-swe-pro",
      "model_slug": "qwen3-8-max",
      "category": "coding",
      "benchmark_name": "SWE-bench Pro",
      "benchmark_version": null,
      "score_numeric": 67.7,
      "score_display": "67.7%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Qwen3.8 official evaluation",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-02T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B",
      "notes": null
    },
    {
      "id": "tq-20260907-qwen38max-terminal21",
      "model_slug": "qwen3-8-max",
      "category": "coding",
      "benchmark_name": "Terminal-Bench 2.1",
      "benchmark_version": null,
      "score_numeric": 86.6,
      "score_display": "86.6%",
      "score_unit": "percent",
      "tools": null,
      "reasoning_effort": null,
      "harness": "Qwen3.8 official evaluation",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-02T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B",
      "notes": null
    },
    {
      "id": "tq-20260907-qwen38max-toolathlon",
      "model_slug": "qwen3-8-max",
      "category": "agentic_computer_use",
      "benchmark_name": "Toolathlon",
      "benchmark_version": "Verified",
      "score_numeric": 72.5,
      "score_display": "72.5%",
      "score_unit": "percent",
      "tools": true,
      "reasoning_effort": null,
      "harness": "Qwen3.8 official evaluation",
      "evaluator": "Alibaba / Qwen",
      "evaluation_date": "2026-08-02T00:00:00.000Z",
      "source": "https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B",
      "notes": null
    }
  ]
}
