{
 "project": "WhichAI",
 "site": "https://whichai.wiki",
 "license": "CC BY 4.0 (attribution: WhichAI, whichai.wiki)",
 "generated": "2026-07-26",
 "snapshot": "July 26, 2026",
 "scaleNote": "Artificial Analysis Intelligence Index, 2026 rebased scale (top models ≈ 55–60). Snapshot: July 24, 2026 (BenchLM mirror), retrieved July 26, 2026.",
 "specNote": "Context, price and release data come from vendor pages and public pricing mirrors (retrieved July 19, 2026). Missing values are not published or not yet verified - shown as -.",
 "count": 108,
 "models": [
  {
   "id": "gpt-6",
   "name": "GPT-6",
   "vendor": "OpenAI",
   "family": null,
   "status": "rumored",
   "access": "Not announced - no confirmed date or specs",
   "tags": [
    "rumored",
    "info-only"
   ],
   "labels": [
    "reasoning",
    "agents"
   ],
   "score": {
    "aa": 65,
    "est": true,
    "cat": {
     "coding": 93,
     "reasoning": 95,
     "writing": 85,
     "agents": 95
    }
   },
   "review": "OpenAI's expected next-numbered flagship: reports point to deeper reasoning, multimodality and agentic tool use, likely in several tiers. Everything about it - including the name - is rumor until OpenAI speaks."
  },
  {
   "id": "gpt-5-6-sol",
   "name": "GPT-5.6 Sol",
   "vendor": "OpenAI",
   "family": "chatgpt",
   "status": "public",
   "access": "Released July 9, 2026 (rolling out) · API $5/$30 per 1M tokens · up to 750 tok/s on Cerebras",
   "tags": [
    "paid",
    "api",
    "prompt-target"
   ],
   "labels": [
    "reasoning",
    "agents",
    "coding"
   ],
   "spec": {
    "released": "2026-07-09",
    "ctx": "1M",
    "priceIn": 5,
    "priceOut": 30,
    "speed": "up to 750 tok/s (Cerebras)"
   },
   "score": {
    "aa": 58.9,
    "est": false,
    "cat": {
     "coding": 91,
     "reasoning": 93,
     "writing": 82,
     "agents": 93
    }
   },
   "review": "OpenAI's new flagship and #1 on the public AA index snapshot: max reasoning effort, programmatic tool calling, and an 'ultra mode' that spins up subagents. The model to beat for hard reasoning and agentic work."
  },
  {
   "id": "gpt-5-6-terra",
   "name": "GPT-5.6 Terra",
   "vendor": "OpenAI",
   "family": "chatgpt",
   "status": "public",
   "spec": {
    "released": "2026-07-09",
    "priceIn": 2.5,
    "priceOut": 15
   },
   "access": "Released July 9, 2026 · API $2.50/$15 per 1M tokens",
   "tags": [
    "paid",
    "api",
    "prompt-target"
   ],
   "labels": [
    "reasoning",
    "value"
   ],
   "score": {
    "aa": 55,
    "est": false,
    "cat": {
     "coding": 86,
     "reasoning": 88,
     "writing": 80,
     "agents": 86
    }
   },
   "review": "The balanced tier of the GPT-5.6 family - GPT-5.5-level quality at half the price. On track to become the everyday ChatGPT default."
  },
  {
   "id": "gpt-5-6-luna",
   "name": "GPT-5.6 Luna",
   "vendor": "OpenAI",
   "family": "chatgpt",
   "status": "public",
   "spec": {
    "released": "2026-07-09",
    "priceIn": 1,
    "priceOut": 6
   },
   "access": "Released July 9, 2026 · API $1/$6 per 1M tokens",
   "tags": [
    "paid",
    "api",
    "prompt-target"
   ],
   "labels": [
    "speed",
    "value"
   ],
   "score": {
    "aa": 51.2,
    "est": false,
    "cat": {
     "coding": 78,
     "reasoning": 80,
     "writing": 74,
     "agents": 78
    }
   },
   "review": "The fast, affordable member of the 5.6 series - near-flagship scores at commodity prices; strong default for high-volume API work."
  },
  {
   "id": "gpt-5-5",
   "name": "GPT-5.5",
   "vendor": "OpenAI",
   "family": "chatgpt",
   "status": "public",
   "spec": {
    "released": "2026-04"
   },
   "access": "Free (limited messages) · Plus $20/mo · Go ~$8/mo · API",
   "tags": [
    "free",
    "paid",
    "api",
    "prompt-target"
   ],
   "labels": [
    "coding",
    "reasoning",
    "writing"
   ],
   "score": {
    "aa": 54.8,
    "est": false,
    "cat": {
     "coding": 88,
     "reasoning": 86,
     "writing": 82,
     "agents": 84
    }
   },
   "review": "The April 2026 flagship, still the strongest broad generalist most people can use free: brainstorming, drafting, frontier coding (SWE-bench Verified 88.7%)."
  },
  {
   "id": "gpt-5-5-pro",
   "name": "GPT-5.5 Pro",
   "vendor": "OpenAI",
   "family": "chatgpt",
   "status": "public",
   "access": "ChatGPT Pro ($200/mo) · API",
   "tags": [
    "paid",
    "api",
    "prompt-target"
   ],
   "labels": [
    "reasoning"
   ],
   "score": {
    "aa": 58,
    "est": true,
    "cat": {
     "coding": 90,
     "reasoning": 91,
     "writing": 82,
     "agents": 87
    }
   },
   "review": "Maximum-reasoning tier of GPT-5.5 - statistically tied with the very top on many boards. Overkill for everyday tasks."
  },
  {
   "id": "gpt-5-4",
   "name": "GPT-5.4",
   "vendor": "OpenAI",
   "family": "chatgpt",
   "status": "public",
   "access": "ChatGPT Plus · API · inside Perplexity Pro",
   "tags": [
    "paid",
    "api",
    "prompt-target"
   ],
   "labels": [
    "coding",
    "reasoning"
   ],
   "score": {
    "aa": 51.4,
    "est": false,
    "cat": {
     "coding": 82,
     "reasoning": 80,
     "writing": 76,
     "agents": 78
    }
   },
   "review": "The previous flagship, still very strong and a reliable known quantity - often cheaper per token than its successors."
  },
  {
   "id": "gpt-5-4-mini",
   "name": "GPT-5.4 mini",
   "vendor": "OpenAI",
   "family": "chatgpt",
   "status": "public",
   "access": "API (cheap tier) · serves ChatGPT free-tier overflow",
   "tags": [
    "free",
    "paid",
    "api",
    "prompt-target"
   ],
   "labels": [
    "speed",
    "value"
   ],
   "score": {
    "aa": 40,
    "est": false,
    "cat": {
     "coding": 66,
     "reasoning": 64,
     "writing": 62,
     "agents": 62
    }
   },
   "review": "Small, quick, cheap - the sensible default for classification, extraction and light drafting at volume."
  },
  {
   "id": "gpt-5-4-nano",
   "name": "GPT-5.4 nano",
   "vendor": "OpenAI",
   "family": "chatgpt",
   "status": "public",
   "access": "API (cheapest OpenAI tier)",
   "tags": [
    "paid",
    "api"
   ],
   "labels": [
    "speed",
    "value"
   ],
   "score": {
    "aa": 38.2,
    "est": false,
    "cat": {
     "coding": 60,
     "reasoning": 58,
     "writing": 56,
     "agents": 55
    }
   },
   "review": "OpenAI's smallest current model - surprisingly capable for its price class, built for massive-scale simple tasks."
  },
  {
   "id": "gpt-5-3-codex",
   "name": "GPT-5.3-Codex",
   "vendor": "OpenAI",
   "family": "chatgpt",
   "status": "public",
   "access": "API · Codex tooling",
   "tags": [
    "paid",
    "api",
    "prompt-target"
   ],
   "labels": [
    "coding",
    "agents"
   ],
   "score": {
    "aa": 44.3,
    "est": false,
    "cat": {
     "coding": 85,
     "reasoning": 72,
     "writing": 55,
     "agents": 84
    }
   },
   "review": "Coding specialist built for agentic dev workflows rather than chat - punches far above its overall index score inside a code editor."
  },
  {
   "id": "gpt-5-2",
   "name": "GPT-5.2",
   "vendor": "OpenAI",
   "family": "chatgpt",
   "status": "legacy",
   "access": "Still serves some free-tier traffic",
   "tags": [
    "free",
    "legacy",
    "prompt-target"
   ],
   "labels": [
    "value"
   ],
   "score": {
    "aa": 42.2,
    "est": false,
    "cat": {
     "coding": 68,
     "reasoning": 66,
     "writing": 66,
     "agents": 62
    }
   },
   "review": "Early-2026 default, now superseded - fine for light tasks, noticeably weaker on hard reasoning."
  },
  {
   "id": "gpt-5-1",
   "name": "GPT-5.1",
   "vendor": "OpenAI",
   "family": "chatgpt",
   "status": "legacy",
   "access": "API (legacy)",
   "tags": [
    "paid",
    "api",
    "legacy"
   ],
   "labels": [],
   "score": {
    "aa": 36.9,
    "est": false,
    "cat": {
     "coding": 62,
     "reasoning": 60,
     "writing": 62,
     "agents": 55
    }
   },
   "review": "Late-2025 generation, kept for compatibility - two clear steps behind the current line."
  },
  {
   "id": "gpt-5",
   "name": "GPT-5",
   "vendor": "OpenAI",
   "family": "chatgpt",
   "status": "legacy",
   "access": "API (legacy)",
   "tags": [
    "paid",
    "api",
    "legacy"
   ],
   "labels": [],
   "score": {
    "aa": 34.7,
    "est": false,
    "cat": {
     "coding": 60,
     "reasoning": 58,
     "writing": 60,
     "agents": 52
    }
   },
   "review": "The model that unified the GPT-4/o-series era in August 2025. Historically important; superseded several times since."
  },
  {
   "id": "o3-pro",
   "name": "o3-pro",
   "vendor": "OpenAI",
   "family": "chatgpt",
   "status": "legacy",
   "access": "API (legacy reasoning line)",
   "tags": [
    "paid",
    "api",
    "legacy"
   ],
   "labels": [
    "reasoning"
   ],
   "score": {
    "aa": 32.5,
    "est": false,
    "cat": {
     "coding": 58,
     "reasoning": 66,
     "writing": 45,
     "agents": 50
    }
   },
   "review": "The 2025 deep-reasoning specialist - slow and dated now, but a milestone in chain-of-thought models."
  },
  {
   "id": "o3",
   "name": "o3",
   "vendor": "OpenAI",
   "family": "chatgpt",
   "status": "legacy",
   "access": "API (legacy)",
   "tags": [
    "paid",
    "api",
    "legacy"
   ],
   "labels": [
    "reasoning"
   ],
   "score": {
    "aa": 30.4,
    "est": false,
    "cat": {
     "coding": 55,
     "reasoning": 62,
     "writing": 42,
     "agents": 48
    }
   },
   "review": "Once the reasoning benchmark-setter; today mainly of historical and compatibility interest."
  },
  {
   "id": "gpt-4-1",
   "name": "GPT-4.1",
   "vendor": "OpenAI",
   "family": "chatgpt",
   "status": "legacy",
   "access": "API (legacy)",
   "tags": [
    "paid",
    "api",
    "legacy"
   ],
   "labels": [
    "long-context"
   ],
   "score": {
    "aa": 19.4,
    "est": false,
    "cat": {
     "coding": 42,
     "reasoning": 35,
     "writing": 42,
     "agents": 30
    }
   },
   "review": "2025 workhorse with a then-notable 1M context. Cheap, stable, far from frontier."
  },
  {
   "id": "gpt-4o",
   "name": "GPT-4o",
   "vendor": "OpenAI",
   "family": "chatgpt",
   "status": "legacy",
   "access": "API (legacy)",
   "tags": [
    "paid",
    "api",
    "legacy"
   ],
   "labels": [
    "vision"
   ],
   "score": {
    "aa": 11.2,
    "est": false,
    "cat": {
     "coding": 28,
     "reasoning": 22,
     "writing": 35,
     "agents": 18
    }
   },
   "review": "The 2024-25 multimodal workhorse. Historically important, no longer competitive on quality - kept for compatibility."
  },
  {
   "id": "gpt-oss-120b",
   "name": "gpt-oss-120b",
   "vendor": "OpenAI",
   "family": "llama",
   "status": "public",
   "access": "Open weights · free/cheap API hosts",
   "tags": [
    "free",
    "open-weights",
    "api",
    "prompt-target"
   ],
   "labels": [
    "local",
    "value"
   ],
   "score": {
    "aa": 23.8,
    "est": false,
    "cat": {
     "coding": 45,
     "reasoning": 44,
     "writing": 40,
     "agents": 38
    }
   },
   "review": "OpenAI's open-weight release - modest by 2026 standards but fully downloadable and permissively licensed."
  },
  {
   "id": "gpt-oss-20b",
   "name": "gpt-oss-20b",
   "vendor": "OpenAI",
   "family": "llama",
   "status": "public",
   "access": "Open weights · runs on a laptop",
   "tags": [
    "free",
    "open-weights",
    "prompt-target"
   ],
   "labels": [
    "local",
    "speed"
   ],
   "score": {
    "aa": 14.9,
    "est": false,
    "cat": {
     "coding": 32,
     "reasoning": 30,
     "writing": 30,
     "agents": 25
    }
   },
   "review": "The small sibling - one of the easiest capable models to run fully offline on consumer hardware."
  },
  {
   "id": "claude-opus-5",
   "name": "Claude Opus 5",
   "vendor": "Anthropic",
   "family": "claude",
   "status": "public",
   "access": "Released July 24, 2026 · Claude Pro / Max plans · API $5/$25 per 1M tokens (Fast mode $10/$50)",
   "tags": [
    "paid",
    "api",
    "prompt-target"
   ],
   "labels": [
    "reasoning",
    "coding",
    "agents",
    "long-context"
   ],
   "spec": {
    "released": "2026-07-24",
    "ctx": "1M",
    "modal": "Text + vision",
    "priceIn": 5,
    "priceOut": 25,
    "note": "cache hits $0.50/M; Batch -50%; 128K output"
   },
   "score": {
    "aa": 60.7,
    "est": false,
    "cat": {
     "coding": 93,
     "reasoning": 94,
     "writing": 88,
     "agents": 95
    }
   },
   "review": "Anthropic's fourth release in two months and the new overall #1: tops the AA Intelligence Index (60.7) and the Agentic Index (55.3) with 1M context, default thinking and computer use, at unchanged Opus pricing. Near-Fable quality at half the price."
  },
  {
   "id": "mythos-5",
   "name": "Claude Mythos 5",
   "vendor": "Anthropic",
   "family": null,
   "status": "private",
   "access": "Approved organizations only (Project Glasswing); US government-cleared, expanding",
   "tags": [
    "private",
    "info-only"
   ],
   "labels": [
    "reasoning",
    "enterprise"
   ],
   "score": {
    "aa": 61,
    "est": true,
    "cat": {
     "coding": 90,
     "reasoning": 94,
     "writing": 88,
     "agents": 90
    }
   },
   "review": "Same underlying model as Fable 5 with dual-use safeguards lifted for vetted organizations; deployed on critical infrastructure in 15+ countries."
  },
  {
   "id": "fable-5",
   "name": "Claude Fable 5",
   "vendor": "Anthropic",
   "family": "claude",
   "status": "public",
   "access": "Claude Pro / Max plans (usage credits since July 7) · API $10/$50 per 1M tokens",
   "spec": {
    "ctx": "1M",
    "modal": "Text + vision",
    "priceIn": 10,
    "priceOut": 50,
    "note": "up to 128K output tokens"
   },
   "tags": [
    "paid",
    "api",
    "prompt-target"
   ],
   "labels": [
    "writing",
    "reasoning",
    "agents"
   ],
   "score": {
    "aa": 59.9,
    "est": false,
    "cat": {
     "coding": 89,
     "reasoning": 92,
     "writing": 93,
     "agents": 90
    }
   },
   "review": "Anthropic's newest flagship (Mythos-class) - tops the AA index at max effort and leads creative-writing boards for voice, subtext and character work; briefly export-controlled, now globally available."
  },
  {
   "id": "opus-4-8",
   "name": "Claude Opus 4.8",
   "vendor": "Anthropic",
   "family": "claude",
   "status": "public",
   "spec": {
    "ctx": "1M",
    "priceIn": 5,
    "priceOut": 25
   },
   "access": "Claude Pro $20/mo · API",
   "tags": [
    "paid",
    "api",
    "prompt-target"
   ],
   "labels": [
    "coding",
    "agents",
    "reasoning"
   ],
   "score": {
    "aa": 55.7,
    "est": false,
    "cat": {
     "coding": 92,
     "reasoning": 88,
     "writing": 84,
     "agents": 91
    }
   },
   "review": "#2 on the public AA snapshot and the leader in real-repo coding (SWE-bench Pro 69.2%). For serious software work, still the default answer."
  },
  {
   "id": "opus-4-7",
   "name": "Claude Opus 4.7",
   "vendor": "Anthropic",
   "family": "claude",
   "status": "legacy",
   "access": "API (being phased down)",
   "tags": [
    "paid",
    "api",
    "legacy",
    "prompt-target"
   ],
   "labels": [
    "coding",
    "reasoning"
   ],
   "score": {
    "aa": 53.5,
    "est": false,
    "cat": {
     "coding": 86,
     "reasoning": 84,
     "writing": 80,
     "agents": 84
    }
   },
   "review": "Still top-10 with adaptive reasoning - a capable fallback where 4.8 isn't available."
  },
  {
   "id": "sonnet-5",
   "name": "Claude Sonnet 5",
   "vendor": "Anthropic",
   "family": "claude",
   "status": "public",
   "spec": {
    "ctx": "1M",
    "priceIn": 2,
    "priceOut": 10,
    "note": "intro API price to Aug 31, 2026 - then $3/$15"
   },
   "access": "FREE tier of claude.ai · API",
   "tags": [
    "free",
    "api",
    "prompt-target"
   ],
   "labels": [
    "writing",
    "coding",
    "value"
   ],
   "score": {
    "aa": 53.4,
    "est": false,
    "cat": {
     "coding": 84,
     "reasoning": 82,
     "writing": 86,
     "agents": 80
    }
   },
   "review": "The strongest free option on the market - frontier-adjacent writing and coding quality at zero cost. Best zero-budget daily driver for text work."
  },
  {
   "id": "sonnet-4-6",
   "name": "Claude Sonnet 4.6",
   "vendor": "Anthropic",
   "family": "claude",
   "status": "public",
   "access": "Claude Pro · API",
   "tags": [
    "paid",
    "api",
    "prompt-target"
   ],
   "labels": [
    "writing",
    "value"
   ],
   "score": {
    "aa": 35.9,
    "est": false,
    "cat": {
     "coding": 68,
     "reasoning": 62,
     "writing": 72,
     "agents": 62
    }
   },
   "review": "The previous balanced default of the Claude line - solid quality at higher limits and lower cost."
  },
  {
   "id": "haiku-4-5",
   "name": "Claude Haiku 4.5",
   "vendor": "Anthropic",
   "family": "claude",
   "status": "public",
   "access": "All Claude tiers · cheapest Claude via API",
   "tags": [
    "free",
    "paid",
    "api",
    "prompt-target"
   ],
   "labels": [
    "speed",
    "value"
   ],
   "score": {
    "aa": 27,
    "est": true,
    "cat": {
     "coding": 52,
     "reasoning": 45,
     "writing": 55,
     "agents": 45
    }
   },
   "review": "Fastest and cheapest of the line - quick answers, classification, high-volume light tasks."
  },
  {
   "id": "opus-4-6",
   "name": "Claude Opus 4.6",
   "vendor": "Anthropic",
   "family": "claude",
   "status": "legacy",
   "access": "API (legacy)",
   "tags": [
    "paid",
    "api",
    "legacy"
   ],
   "labels": [
    "coding"
   ],
   "score": {
    "aa": 43.7,
    "est": false,
    "cat": {
     "coding": 76,
     "reasoning": 74,
     "writing": 72,
     "agents": 72
    }
   },
   "review": "Early-2026 Opus, superseded twice since - retained mainly for pinned enterprise workflows."
  },
  {
   "id": "opus-4-5",
   "name": "Claude Opus 4.5",
   "vendor": "Anthropic",
   "family": "claude",
   "status": "legacy",
   "access": "API (legacy)",
   "tags": [
    "paid",
    "api",
    "legacy"
   ],
   "labels": [
    "coding"
   ],
   "score": {
    "aa": 40.8,
    "est": false,
    "cat": {
     "coding": 72,
     "reasoning": 70,
     "writing": 70,
     "agents": 68
    }
   },
   "review": "The late-2025 flagship that made agentic coding mainstream; now two generations back."
  },
  {
   "id": "claude-4-1-opus",
   "name": "Claude Opus 4.1",
   "vendor": "Anthropic",
   "family": "claude",
   "status": "legacy",
   "access": "API (legacy)",
   "tags": [
    "paid",
    "api",
    "legacy"
   ],
   "labels": [],
   "score": {
    "aa": 33.7,
    "est": false,
    "cat": {
     "coding": 62,
     "reasoning": 60,
     "writing": 64,
     "agents": 55
    }
   },
   "review": "Mid-2025 generation - of historical interest and for compatibility with older pipelines."
  },
  {
   "id": "claude-3-opus",
   "name": "Claude 3 Opus",
   "vendor": "Anthropic",
   "family": "claude",
   "status": "legacy",
   "access": "API (legacy)",
   "tags": [
    "paid",
    "api",
    "legacy"
   ],
   "labels": [
    "writing"
   ],
   "score": {
    "aa": 11.8,
    "est": false,
    "cat": {
     "coding": 25,
     "reasoning": 22,
     "writing": 40,
     "agents": 15
    }
   },
   "review": "The 2024 model that first made Claude famous for prose quality. Retired from frontier duty."
  },
  {
   "id": "gemini-3-ultra",
   "name": "Gemini 3 Ultra",
   "vendor": "Google",
   "family": null,
   "status": "rumored",
   "access": "Reported as upcoming - no official availability",
   "tags": [
    "rumored",
    "info-only"
   ],
   "labels": [
    "reasoning",
    "long-context"
   ],
   "score": {
    "aa": 56,
    "est": true,
    "cat": {
     "coding": 86,
     "reasoning": 88,
     "writing": 80,
     "agents": 85
    }
   },
   "review": "Reported top tier above Gemini 3.1 Pro with wider API access said to be coming; Google has published nothing definitive - treat specs as rumor."
  },
  {
   "id": "gemini-3-1-pro",
   "name": "Gemini 3.1 Pro",
   "vendor": "Google",
   "family": "gemini",
   "status": "public",
   "spec": {
    "ctx": "1M",
    "priceIn": 2,
    "priceOut": 12,
    "note": "$4/$18 above 200K input tokens"
   },
   "access": "Google AI Pro $19.99/mo · API · inside Perplexity Pro",
   "tags": [
    "paid",
    "api",
    "prompt-target"
   ],
   "labels": [
    "coding",
    "long-context",
    "vision"
   ],
   "score": {
    "aa": 46.5,
    "est": false,
    "cat": {
     "coding": 84,
     "reasoning": 78,
     "writing": 74,
     "agents": 78
    }
   },
   "review": "#1 in LMArena's coding arena at launch, deep Google Workspace integration and strong multimodal work."
  },
  {
   "id": "gemini-3-6-flash",
   "name": "Gemini 3.6 Flash",
   "vendor": "Google",
   "family": "gemini",
   "status": "public",
   "access": "Free tier in AI Studio and the Gemini app (rate-limited) · API $1.50/$7.50 per 1M tokens",
   "tags": [
    "free",
    "paid",
    "api",
    "prompt-target",
    "auto-run"
   ],
   "labels": [
    "speed",
    "value"
   ],
   "spec": {
    "released": "2026-07-21",
    "ctx": "1M",
    "priceIn": 1.5,
    "priceOut": 7.5
   },
   "score": {
    "aa": 50.1,
    "est": false,
    "cat": {
     "coding": 76,
     "reasoning": 78,
     "writing": 72,
     "agents": 76
    }
   },
   "review": "Google's July refresh of the free workhorse: 3.5 Flash quality with cheaper output ($7.50 vs $9), 1M context, day one in AI Studio, the API and the app. The new default of the Gemini line."
  },
  {
   "id": "gemini-3-5-flash-lite",
   "name": "Gemini 3.5 Flash-Lite",
   "vendor": "Google",
   "family": "gemini",
   "status": "public",
   "access": "API $0.30/$2.50 per 1M tokens",
   "tags": [
    "paid",
    "api",
    "prompt-target"
   ],
   "labels": [
    "speed",
    "value"
   ],
   "spec": {
    "released": "2026-07-21",
    "priceIn": 0.3,
    "priceOut": 2.5
   },
   "score": {
    "aa": 36.5,
    "est": false,
    "cat": {
     "coding": 58,
     "reasoning": 60,
     "writing": 56,
     "agents": 55
    }
   },
   "review": "The routing tier of the July refresh: classification, extraction, tagging and short summaries at commodity prices. Not built for hard reasoning."
  },
  {
   "id": "gemini-3-5-flash",
   "name": "Gemini 3.5 Flash",
   "vendor": "Google",
   "family": "gemini",
   "status": "public",
   "spec": {
    "ctx": "1M",
    "priceIn": 1.5,
    "priceOut": 9
   },
   "access": "FREE (most generous free tier) · free API via Google AI Studio",
   "tags": [
    "free",
    "api",
    "prompt-target",
    "auto-run"
   ],
   "labels": [
    "speed",
    "value",
    "vision"
   ],
   "score": {
    "aa": 50.2,
    "est": false,
    "cat": {
     "coding": 80,
     "reasoning": 78,
     "writing": 74,
     "agents": 76
    }
   },
   "review": "Remarkable: the newest Flash now outscores the older 3.1 Pro on the AA index - frontier-adjacent quality, free, and the model WhichAI auto-runs with your AI Studio key."
  },
  {
   "id": "gemini-3-pro",
   "name": "Gemini 3 Pro",
   "vendor": "Google",
   "family": "gemini",
   "status": "public",
   "access": "Google AI Pro · API",
   "tags": [
    "paid",
    "api",
    "prompt-target"
   ],
   "labels": [
    "vision"
   ],
   "score": {
    "aa": 39.5,
    "est": false,
    "cat": {
     "coding": 70,
     "reasoning": 66,
     "writing": 66,
     "agents": 66
    }
   },
   "review": "Previous Pro generation, still widely deployed across Google products."
  },
  {
   "id": "gemini-3-flash",
   "name": "Gemini 3 Flash",
   "vendor": "Google",
   "family": "gemini",
   "status": "public",
   "access": "Free tier · cheap API",
   "tags": [
    "free",
    "api",
    "prompt-target"
   ],
   "labels": [
    "speed",
    "value"
   ],
   "score": {
    "aa": 27.4,
    "est": false,
    "cat": {
     "coding": 52,
     "reasoning": 48,
     "writing": 50,
     "agents": 46
    }
   },
   "review": "The earlier fast tier - inexpensive and quick, now clearly behind 3.5 Flash."
  },
  {
   "id": "gemini-3-1-flash-lite",
   "name": "Gemini 3.1 Flash-Lite",
   "vendor": "Google",
   "family": "gemini",
   "status": "public",
   "access": "Free API (highest free rate limits: 15 RPM, 1,000/day)",
   "tags": [
    "free",
    "api",
    "prompt-target"
   ],
   "labels": [
    "speed",
    "value"
   ],
   "score": {
    "aa": 25,
    "est": false,
    "cat": {
     "coding": 48,
     "reasoning": 44,
     "writing": 46,
     "agents": 42
    }
   },
   "review": "The volume workhorse: the most generous free API quota of any capable model."
  },
  {
   "id": "gemini-2-5-pro",
   "name": "Gemini 2.5 Pro",
   "vendor": "Google",
   "family": "gemini",
   "status": "legacy",
   "access": "API (legacy, paid-only since April 2026)",
   "tags": [
    "paid",
    "api",
    "legacy"
   ],
   "labels": [
    "long-context"
   ],
   "score": {
    "aa": 25.8,
    "est": false,
    "cat": {
     "coding": 50,
     "reasoning": 48,
     "writing": 48,
     "agents": 44
    }
   },
   "review": "The 2025 flagship, superseded by the Gemini 3 line."
  },
  {
   "id": "gemini-2-5-flash",
   "name": "Gemini 2.5 Flash",
   "vendor": "Google",
   "family": "gemini",
   "status": "legacy",
   "access": "API (legacy)",
   "tags": [
    "paid",
    "api",
    "legacy"
   ],
   "labels": [
    "speed"
   ],
   "score": {
    "aa": 14.1,
    "est": false,
    "cat": {
     "coding": 32,
     "reasoning": 28,
     "writing": 32,
     "agents": 26
    }
   },
   "review": "2025's fast tier, kept for compatibility."
  },
  {
   "id": "gemma-4-31b",
   "name": "Gemma 4 31B",
   "vendor": "Google",
   "family": "llama",
   "status": "public",
   "access": "Open weights · runs on consumer hardware",
   "tags": [
    "free",
    "open-weights",
    "prompt-target"
   ],
   "labels": [
    "local",
    "value"
   ],
   "score": {
    "aa": 29.4,
    "est": false,
    "cat": {
     "coding": 54,
     "reasoning": 52,
     "writing": 54,
     "agents": 46
    }
   },
   "review": "Google's open family - one of the best models you can run locally on a single GPU."
  },
  {
   "id": "gemma-4-12b",
   "name": "Gemma 4 12B",
   "vendor": "Google",
   "family": "llama",
   "status": "public",
   "access": "Open weights · laptop-friendly",
   "tags": [
    "free",
    "open-weights",
    "prompt-target"
   ],
   "labels": [
    "local",
    "speed"
   ],
   "score": {
    "aa": 21.8,
    "est": false,
    "cat": {
     "coding": 42,
     "reasoning": 40,
     "writing": 44,
     "agents": 34
    }
   },
   "review": "The mid-size Gemma - a favorite for private, offline use on ordinary hardware."
  },
  {
   "id": "gemma-3-27b",
   "name": "Gemma 3 27B",
   "vendor": "Google",
   "family": "llama",
   "status": "legacy",
   "access": "Open weights",
   "tags": [
    "free",
    "open-weights",
    "legacy"
   ],
   "labels": [
    "local"
   ],
   "score": {
    "aa": 7.4,
    "est": false,
    "cat": {
     "coding": 20,
     "reasoning": 16,
     "writing": 24,
     "agents": 12
    }
   },
   "review": "The 2025 open generation, superseded by Gemma 4."
  },
  {
   "id": "grok-5",
   "name": "Grok 5",
   "vendor": "xAI",
   "family": null,
   "status": "rumored",
   "access": "Not released - repeatedly delayed (prediction markets price a near-term launch low)",
   "tags": [
    "rumored",
    "info-only"
   ],
   "labels": [
    "reasoning"
   ],
   "score": {
    "aa": 58,
    "est": true,
    "cat": {
     "coding": 88,
     "reasoning": 90,
     "writing": 78,
     "agents": 88
    }
   },
   "review": "xAI's long-promised next flagship. Timelines have slipped more than once; any spec you read is speculation."
  },
  {
   "id": "grok-4-5",
   "name": "Grok 4.5",
   "vendor": "xAI",
   "family": "grok",
   "status": "public",
   "spec": {
    "released": "2026-07-08",
    "priceIn": 2,
    "priceOut": 6,
    "speed": "~80 tok/s, high token efficiency"
   },
   "access": "Released July 8, 2026 · API $2/$6 per 1M tokens · in Grok Build & Cursor · not yet in the EU",
   "tags": [
    "paid",
    "api",
    "prompt-target"
   ],
   "labels": [
    "coding",
    "agents",
    "value"
   ],
   "score": {
    "aa": 53.8,
    "est": false,
    "cat": {
     "coding": 89,
     "reasoning": 84,
     "writing": 74,
     "agents": 90
    }
   },
   "review": "xAI's first model built specifically for coding and agents (1.5T-parameter base, trained on real Cursor sessions): #4 on the AA index at a fraction of flagship prices. EU availability pending."
  },
  {
   "id": "grok-4-1",
   "name": "Grok 4.1",
   "vendor": "xAI",
   "family": "grok",
   "status": "public",
   "access": "Free tier on X / grok.com · paid from ~$16/mo · API",
   "tags": [
    "free",
    "paid",
    "api",
    "prompt-target"
   ],
   "labels": [
    "research",
    "long-context"
   ],
   "score": {
    "aa": 36,
    "est": true,
    "cat": {
     "coding": 62,
     "reasoning": 66,
     "writing": 62,
     "agents": 62
    }
   },
   "review": "Direct, up-to-the-minute answers with X data, a Think mode and a 2M-token context; among the lowest measured hallucination rates (~4%)."
  },
  {
   "id": "grok-4-3",
   "name": "Grok 4.3",
   "vendor": "xAI",
   "family": "grok",
   "status": "public",
   "access": "grok.com paid tiers · API",
   "tags": [
    "paid",
    "api",
    "prompt-target"
   ],
   "labels": [
    "reasoning"
   ],
   "score": {
    "aa": 37.6,
    "est": false,
    "cat": {
     "coding": 66,
     "reasoning": 68,
     "writing": 60,
     "agents": 64
    }
   },
   "review": "Spring-2026 generation - a solid step over Grok 4.1 on reasoning, now overshadowed by 4.5."
  },
  {
   "id": "grok-build-0-1",
   "name": "Grok Build 0.1",
   "vendor": "xAI",
   "family": "grok",
   "status": "public",
   "access": "Inside Grok Build (xAI's coding agent/CLI)",
   "tags": [
    "paid",
    "prompt-target"
   ],
   "labels": [
    "agents",
    "coding"
   ],
   "score": {
    "aa": 39.8,
    "est": false,
    "cat": {
     "coding": 76,
     "reasoning": 62,
     "writing": 45,
     "agents": 80
    }
   },
   "review": "The agent-tuned model behind xAI's Grok Build environment - sessions, skills, plugins and repo-scale work rather than chat."
  },
  {
   "id": "grok-code-fast-1",
   "name": "Grok Code Fast 1",
   "vendor": "xAI",
   "family": "grok",
   "status": "public",
   "access": "API (very cheap) · common inside IDE integrations",
   "tags": [
    "paid",
    "api",
    "prompt-target"
   ],
   "labels": [
    "coding",
    "speed",
    "value"
   ],
   "score": {
    "aa": 21.6,
    "est": false,
    "cat": {
     "coding": 55,
     "reasoning": 35,
     "writing": 25,
     "agents": 48
    }
   },
   "review": "Tiny, fast coding model - autocompletion-class work at near-zero cost, not for hard problems."
  },
  {
   "id": "grok-4",
   "name": "Grok 4",
   "vendor": "xAI",
   "family": "grok",
   "status": "legacy",
   "access": "API (legacy)",
   "tags": [
    "paid",
    "api",
    "legacy"
   ],
   "labels": [],
   "score": {
    "aa": 33.3,
    "est": false,
    "cat": {
     "coding": 58,
     "reasoning": 62,
     "writing": 52,
     "agents": 56
    }
   },
   "review": "Previous xAI generation - superseded by 4.1's larger context and then by the 4.3/4.5 wave."
  },
  {
   "id": "llama-5",
   "name": "Llama 5",
   "vendor": "Meta",
   "family": null,
   "status": "rumored",
   "access": "Not announced - no weights, no model card",
   "tags": [
    "rumored",
    "info-only"
   ],
   "labels": [
    "local"
   ],
   "score": {
    "aa": 45,
    "est": true,
    "cat": {
     "coding": 74,
     "reasoning": 72,
     "writing": 68,
     "agents": 70
    }
   },
   "review": "Persistent rumors, zero official signal: no Meta announcement and no Hugging Face weights. Any 'Llama 5 review' online is premature or invented. Meta's proprietary Muse line now gets the frontier attention."
  },
  {
   "id": "muse-spark-1-1",
   "name": "Muse Spark 1.1",
   "vendor": "Meta",
   "family": "meta",
   "status": "public",
   "spec": {
    "released": "2026-07-09"
   },
   "access": "API via Meta Superintelligence Labs (Meta's first paid model, July 9, 2026)",
   "tags": [
    "paid",
    "api",
    "prompt-target"
   ],
   "labels": [
    "agents",
    "vision",
    "reasoning"
   ],
   "score": {
    "aa": 50.6,
    "est": false,
    "cat": {
     "coding": 74,
     "reasoning": 76,
     "writing": 66,
     "agents": 80
    }
   },
   "review": "Meta's first proprietary, paid model line (Meta Superintelligence Labs): natively multimodal reasoning with visual chain-of-thought and multi-agent orchestration. A strategic U-turn from open weights."
  },
  {
   "id": "meta-ai",
   "name": "Meta AI",
   "vendor": "Meta",
   "family": "meta",
   "status": "public",
   "access": "FREE - meta.ai and inside WhatsApp, Instagram, Messenger",
   "tags": [
    "free",
    "prompt-target"
   ],
   "labels": [
    "value"
   ],
   "score": {
    "aa": null,
    "est": true,
    "cat": null
   },
   "review": "Not a model but an assistant woven into Meta's apps, now increasingly powered by the Muse line with real-time info via search. Ubiquitous and free."
  },
  {
   "id": "llama-4-maverick",
   "name": "Llama 4 Maverick",
   "vendor": "Meta",
   "family": "llama",
   "status": "public",
   "access": "Open weights · FREE fast inference on Groq",
   "tags": [
    "free",
    "open-weights",
    "api",
    "prompt-target",
    "auto-run"
   ],
   "labels": [
    "local",
    "speed",
    "value"
   ],
   "score": {
    "aa": 14.3,
    "est": false,
    "cat": {
     "coding": 34,
     "reasoning": 30,
     "writing": 36,
     "agents": 26
    }
   },
   "review": "The larger Llama 4 - near-instant on Groq and fine for drafts, but the index is honest: it sits far below the 2026 open leaders."
  },
  {
   "id": "llama-4-scout",
   "name": "Llama 4 Scout",
   "vendor": "Meta",
   "family": "llama",
   "status": "public",
   "access": "Open weights · very long context",
   "tags": [
    "free",
    "open-weights",
    "prompt-target"
   ],
   "labels": [
    "local",
    "long-context"
   ],
   "score": {
    "aa": 10,
    "est": false,
    "cat": {
     "coding": 26,
     "reasoning": 22,
     "writing": 30,
     "agents": 20
    }
   },
   "review": "Llama 4's efficient sibling with an enormous context window - useful locally, weak on hard tasks."
  },
  {
   "id": "llama-3-3-70b",
   "name": "Llama 3.3 70B",
   "vendor": "Meta",
   "family": "llama",
   "status": "legacy",
   "access": "Open weights · FREE on Groq (WhichAI default runner)",
   "tags": [
    "free",
    "open-weights",
    "api",
    "legacy",
    "prompt-target",
    "auto-run"
   ],
   "labels": [
    "local",
    "speed"
   ],
   "score": {
    "aa": 7,
    "est": true,
    "cat": {
     "coding": 20,
     "reasoning": 16,
     "writing": 24,
     "agents": 14
    }
   },
   "review": "Older but dependable - the default model WhichAI auto-runs with a free Groq key."
  },
  {
   "id": "llama-3-1-405b",
   "name": "Llama 3.1 405B",
   "vendor": "Meta",
   "family": "llama",
   "status": "legacy",
   "access": "Open weights",
   "tags": [
    "free",
    "open-weights",
    "legacy"
   ],
   "labels": [
    "local"
   ],
   "score": {
    "aa": 8.5,
    "est": false,
    "cat": {
     "coding": 22,
     "reasoning": 20,
     "writing": 26,
     "agents": 16
    }
   },
   "review": "2024's open giant - a milestone for open AI, now outclassed by far smaller models."
  },
  {
   "id": "copilot",
   "name": "Copilot (GPT-5 family)",
   "vendor": "Microsoft",
   "family": "copilot",
   "status": "public",
   "access": "Free tier · Copilot Pro $20/mo · built into Windows & Microsoft 365",
   "tags": [
    "free",
    "paid",
    "prompt-target"
   ],
   "labels": [
    "enterprise"
   ],
   "score": {
    "aa": null,
    "est": true,
    "cat": null
   },
   "review": "Not a model but a delivery vehicle: OpenAI GPT-5-family models woven into Word, Excel, PowerPoint, Teams and Windows."
  },
  {
   "id": "phi-4",
   "name": "Phi-4",
   "vendor": "Microsoft",
   "family": "llama",
   "status": "public",
   "access": "Open weights · laptop-friendly",
   "tags": [
    "free",
    "open-weights",
    "prompt-target"
   ],
   "labels": [
    "local",
    "value"
   ],
   "score": {
    "aa": 4.9,
    "est": false,
    "cat": {
     "coding": 15,
     "reasoning": 14,
     "writing": 16,
     "agents": 8
    }
   },
   "review": "Microsoft's small open research line - historically interesting for data-quality tricks, no longer competitive."
  },
  {
   "id": "sonar",
   "name": "Perplexity Sonar",
   "vendor": "Perplexity AI",
   "family": "perplexity",
   "status": "public",
   "access": "FREE: unlimited basic cited searches + 5 Pro searches/day",
   "tags": [
    "free",
    "api",
    "prompt-target"
   ],
   "labels": [
    "research",
    "speed"
   ],
   "score": {
    "aa": null,
    "est": true,
    "cat": null
   },
   "review": "Search-answer engine rather than a chat model: best-in-class citation accuracy - a smarter, cited replacement for googling."
  },
  {
   "id": "sonar-pro",
   "name": "Perplexity Sonar Pro + Deep Research",
   "vendor": "Perplexity AI",
   "family": "perplexity",
   "status": "public",
   "access": "Perplexity Pro $20/mo (includes GPT, Claude and Gemini flagships inside)",
   "tags": [
    "paid",
    "api",
    "prompt-target"
   ],
   "labels": [
    "research",
    "long-context"
   ],
   "score": {
    "aa": null,
    "est": true,
    "cat": null
   },
   "review": "Deeper retrieval, 20 Deep Research runs/day, and frontier models selectable inside one subscription."
  },
  {
   "id": "deepseek-v4-pro",
   "name": "DeepSeek V4 Pro",
   "vendor": "DeepSeek",
   "family": "deepseek",
   "status": "public",
   "access": "Free chat app · open weights · very cheap API · 1M context default",
   "tags": [
    "free",
    "open-weights",
    "api",
    "prompt-target"
   ],
   "labels": [
    "reasoning",
    "coding",
    "value"
   ],
   "score": {
    "aa": 44.3,
    "est": false,
    "cat": {
     "coding": 87,
     "reasoning": 82,
     "writing": 68,
     "agents": 76
    }
   },
   "review": "The open reasoning champion (1.6T sparse MoE, 49B active): tops open coding boards at famously low prices - the most popular frontier-adjacent free model worldwide."
  },
  {
   "id": "deepseek-v4-flash",
   "name": "DeepSeek V4 Flash",
   "vendor": "DeepSeek",
   "family": "deepseek",
   "status": "public",
   "access": "Free chat app · open weights · cheapest DeepSeek API",
   "tags": [
    "free",
    "open-weights",
    "api",
    "prompt-target"
   ],
   "labels": [
    "speed",
    "value"
   ],
   "score": {
    "aa": 40.3,
    "est": false,
    "cat": {
     "coding": 74,
     "reasoning": 70,
     "writing": 62,
     "agents": 66
    }
   },
   "review": "The efficient V4 (284B total, 13B active) - most of Pro's quality at a fraction of the compute."
  },
  {
   "id": "deepseek-v3-2",
   "name": "DeepSeek V3.2",
   "vendor": "DeepSeek",
   "family": "deepseek",
   "status": "legacy",
   "access": "Open weights · cheap API",
   "tags": [
    "free",
    "open-weights",
    "api",
    "legacy"
   ],
   "labels": [
    "value"
   ],
   "score": {
    "aa": 24.7,
    "est": false,
    "cat": {
     "coding": 50,
     "reasoning": 46,
     "writing": 42,
     "agents": 40
    }
   },
   "review": "The 2025 generation that made ultra-cheap frontier-adjacent AI mainstream; superseded by V4."
  },
  {
   "id": "deepseek-r1",
   "name": "DeepSeek R1",
   "vendor": "DeepSeek",
   "family": "deepseek",
   "status": "legacy",
   "access": "Open weights",
   "tags": [
    "free",
    "open-weights",
    "legacy"
   ],
   "labels": [
    "reasoning"
   ],
   "score": {
    "aa": 20.1,
    "est": false,
    "cat": {
     "coding": 42,
     "reasoning": 50,
     "writing": 35,
     "agents": 32
    }
   },
   "review": "The January 2025 release that shocked the market by open-sourcing o1-class reasoning. Historic; superseded."
  },
  {
   "id": "qwen-3-8-max-preview",
   "name": "Qwen3.8 Max (preview)",
   "vendor": "Alibaba",
   "family": "qwen",
   "status": "preview",
   "access": "Early preview via Qwen chat and Alibaba Cloud (July 19) · specs not final",
   "tags": [
    "preview",
    "api",
    "prompt-target"
   ],
   "labels": [
    "reasoning",
    "multilingual"
   ],
   "spec": {
    "released": "2026-07-19"
   },
   "score": {
    "aa": 48,
    "est": true,
    "cat": {
     "coding": 80,
     "reasoning": 82,
     "writing": 74,
     "agents": 78
    }
   },
   "review": "Alibaba's next flagship in early preview. No published index score yet; expect it above Qwen3.7 Max (46.0) once measured. Everything here is provisional."
  },
  {
   "id": "qwen-3-7-max",
   "name": "Qwen3.7 Max",
   "vendor": "Alibaba",
   "family": "qwen",
   "status": "public",
   "access": "Free chat app · API (closed weights for Max tier)",
   "tags": [
    "free",
    "paid",
    "api",
    "prompt-target"
   ],
   "labels": [
    "multilingual",
    "coding",
    "reasoning"
   ],
   "score": {
    "aa": 46,
    "est": false,
    "cat": {
     "coding": 82,
     "reasoning": 78,
     "writing": 70,
     "agents": 76
    }
   },
   "review": "Alibaba's strongest current model and the top Chinese closed-weights entry on the AA index - outstanding multilingual coverage."
  },
  {
   "id": "qwen-3-7-plus",
   "name": "Qwen3.7 Plus",
   "vendor": "Alibaba",
   "family": "qwen",
   "status": "public",
   "access": "Free chat app · cheap API",
   "tags": [
    "free",
    "api",
    "prompt-target"
   ],
   "labels": [
    "multilingual",
    "value"
   ],
   "score": {
    "aa": 39,
    "est": false,
    "cat": {
     "coding": 70,
     "reasoning": 66,
     "writing": 62,
     "agents": 64
    }
   },
   "review": "The balanced tier of the 3.7 line - strong value for multilingual work at volume."
  },
  {
   "id": "qwen-3-6-plus",
   "name": "Qwen3.6 Plus",
   "vendor": "Alibaba",
   "family": "qwen",
   "status": "public",
   "access": "Free chat app · open-weight siblings · API",
   "tags": [
    "free",
    "open-weights",
    "api",
    "prompt-target"
   ],
   "labels": [
    "multilingual",
    "coding"
   ],
   "score": {
    "aa": 39.6,
    "est": false,
    "cat": {
     "coding": 72,
     "reasoning": 66,
     "writing": 62,
     "agents": 64
    }
   },
   "review": "Among the strongest open lines for coding and clearly the best multilingual open family."
  },
  {
   "id": "qwen-3-6-27b",
   "name": "Qwen3.6-27B",
   "vendor": "Alibaba",
   "family": "qwen",
   "status": "public",
   "access": "Open weights (Apache 2.0) · runs on a good desktop",
   "tags": [
    "free",
    "open-weights",
    "prompt-target"
   ],
   "labels": [
    "local",
    "multilingual",
    "value"
   ],
   "score": {
    "aa": 37,
    "est": false,
    "cat": {
     "coding": 66,
     "reasoning": 62,
     "writing": 58,
     "agents": 58
    }
   },
   "review": "Possibly the best quality-per-GB you can download today - near mid-frontier scores from a 27B you can run yourself."
  },
  {
   "id": "qwen-3-5-397b",
   "name": "Qwen3.5 397B",
   "vendor": "Alibaba",
   "family": "qwen",
   "status": "public",
   "access": "Open weights (Apache 2.0)",
   "tags": [
    "free",
    "open-weights",
    "prompt-target"
   ],
   "labels": [
    "multilingual",
    "reasoning"
   ],
   "score": {
    "aa": 33.7,
    "est": false,
    "cat": {
     "coding": 64,
     "reasoning": 62,
     "writing": 56,
     "agents": 56
    }
   },
   "review": "The open MoE flagship of the 3.5 wave - big, capable, Apache-licensed."
  },
  {
   "id": "qwen3-coder",
   "name": "Qwen3 Coder",
   "vendor": "Alibaba",
   "family": "qwen",
   "status": "public",
   "spec": {
    "ctx": "1M"
   },
   "access": "Open weights · FREE on OpenRouter (qwen/qwen3-coder:free, 1M context) - WhichAI auto-runs it with a free OpenRouter key",
   "tags": [
    "free",
    "open-weights",
    "api",
    "prompt-target",
    "auto-run"
   ],
   "labels": [
    "coding",
    "long-context",
    "value"
   ],
   "score": {
    "aa": 30,
    "est": true,
    "cat": {
     "coding": 68,
     "reasoning": 52,
     "writing": 44,
     "agents": 58
    }
   },
   "review": "The strongest free coding route on OpenRouter right now - 1M context, tool use, zero cost within the daily free limits."
  },
  {
   "id": "kimi-k3",
   "name": "Kimi K3",
   "vendor": "Moonshot AI",
   "family": "kimi",
   "status": "public",
   "access": "API $3/$15 per 1M tokens ($0.30 cached) · kimi.com chat · open weights announced for July 27 (Modified MIT)",
   "tags": [
    "paid",
    "api",
    "prompt-target"
   ],
   "labels": [
    "coding",
    "reasoning",
    "agents",
    "vision",
    "long-context"
   ],
   "spec": {
    "released": "2026-07-16",
    "ctx": "1M",
    "modal": "Text + vision",
    "priceIn": 3,
    "priceOut": 15
   },
   "score": {
    "aa": 57.1,
    "est": false,
    "cat": {
     "coding": 90,
     "reasoning": 91,
     "writing": 78,
     "agents": 94
    }
   },
   "review": "Moonshot's July 16 flagship: 2.8T-parameter MoE with 1M context and native vision. #3 on the AA index (57.1), best published BrowseComp (91.2%) and #1 Frontend Code Arena at launch - the strongest agentic challenger to the US frontier."
  },
  {
   "id": "kimi-k2-6",
   "name": "Kimi K2.6",
   "vendor": "Moonshot AI",
   "family": "kimi",
   "status": "public",
   "access": "Open weights · API hosts · free chat app",
   "tags": [
    "free",
    "open-weights",
    "api",
    "prompt-target"
   ],
   "labels": [
    "coding",
    "agents",
    "long-context"
   ],
   "score": {
    "aa": 44.2,
    "est": false,
    "cat": {
     "coding": 88,
     "reasoning": 76,
     "writing": 66,
     "agents": 82
    }
   },
   "review": "Strong open coding (SWE-bench Verified 80.2%) and long agentic tasks - now the smaller sibling of the flagship K3."
  },
  {
   "id": "kimi-k2-7-code",
   "name": "Kimi K2.7 Code",
   "vendor": "Moonshot AI",
   "family": "kimi",
   "status": "preview",
   "access": "Early access via Moonshot · open weights promised",
   "tags": [
    "preview",
    "open-weights",
    "info-only"
   ],
   "labels": [
    "coding",
    "agents"
   ],
   "score": {
    "aa": 42,
    "est": false,
    "cat": {
     "coding": 86,
     "reasoning": 70,
     "writing": 55,
     "agents": 82
    }
   },
   "review": "Moonshot's agentic-coding bet - early numbers put it just behind K2.6 overall but ahead on real-repo work."
  },
  {
   "id": "kimi-k2-5",
   "name": "Kimi K2.5",
   "vendor": "Moonshot AI",
   "family": "kimi",
   "status": "public",
   "access": "Open weights · API hosts",
   "tags": [
    "free",
    "open-weights",
    "api",
    "prompt-target"
   ],
   "labels": [
    "long-context"
   ],
   "score": {
    "aa": 35.4,
    "est": false,
    "cat": {
     "coding": 72,
     "reasoning": 66,
     "writing": 60,
     "agents": 68
    }
   },
   "review": "The previous Kimi - still a solid open all-rounder with excellent long-context behavior."
  },
  {
   "id": "glm-5-2",
   "name": "GLM-5.2",
   "vendor": "Z.ai",
   "family": "glm",
   "status": "public",
   "access": "Open weights · low-cost API hosts",
   "tags": [
    "free",
    "open-weights",
    "api",
    "prompt-target"
   ],
   "labels": [
    "coding",
    "agents",
    "value"
   ],
   "score": {
    "aa": 51.1,
    "est": false,
    "cat": {
     "coding": 90,
     "reasoning": 80,
     "writing": 68,
     "agents": 84
    }
   },
   "review": "The strongest open-weight model in the world right now (#10 overall, above Gemini 3.1 Pro): SWE-bench Pro 62.1%, #2 React on Code Arena - frontier coding at open prices."
  },
  {
   "id": "glm-5-1",
   "name": "GLM-5.1",
   "vendor": "Z.ai",
   "family": "glm",
   "status": "public",
   "access": "Open weights",
   "tags": [
    "free",
    "open-weights",
    "prompt-target"
   ],
   "labels": [
    "coding"
   ],
   "score": {
    "aa": 40.2,
    "est": false,
    "cat": {
     "coding": 78,
     "reasoning": 68,
     "writing": 60,
     "agents": 72
    }
   },
   "review": "The previous Z.ai release - still strong, quickly superseded by 5.2."
  },
  {
   "id": "glm-5",
   "name": "GLM-5",
   "vendor": "Z.ai",
   "family": "glm",
   "status": "public",
   "access": "Open weights",
   "tags": [
    "free",
    "open-weights",
    "prompt-target"
   ],
   "labels": [
    "coding"
   ],
   "score": {
    "aa": 39.5,
    "est": false,
    "cat": {
     "coding": 76,
     "reasoning": 66,
     "writing": 58,
     "agents": 70
    }
   },
   "review": "The early-2026 base of the GLM-5 family that put Z.ai on the frontier map."
  },
  {
   "id": "hy3",
   "name": "Hunyuan 3.0 (Hy3)",
   "vendor": "Tencent",
   "family": "llama",
   "status": "public",
   "access": "Open weights (Apache 2.0, released July 6, 2026) · 295B MoE, 21B active · 256K context",
   "tags": [
    "free",
    "open-weights",
    "prompt-target"
   ],
   "labels": [
    "multilingual",
    "long-context"
   ],
   "score": {
    "aa": 41.2,
    "est": false,
    "cat": {
     "coding": 62,
     "reasoning": 60,
     "writing": 58,
     "agents": 56
    }
   },
   "review": "Tencent's move into the open-weight camp - one of the last big closed Chinese labs to open up, days ago. Solid mid-frontier scores and a permissive license."
  },
  {
   "id": "doubao-seed-2-1-pro",
   "name": "Doubao Seed 2.1 Pro",
   "vendor": "ByteDance",
   "family": null,
   "status": "public",
   "access": "China-focused: served via ByteDance's Volcano Engine cloud only",
   "tags": [
    "paid",
    "api",
    "info-only"
   ],
   "labels": [
    "enterprise",
    "multilingual"
   ],
   "score": {
    "aa": 42,
    "est": true,
    "cat": {
     "coding": 72,
     "reasoning": 70,
     "writing": 64,
     "agents": 68
    }
   },
   "review": "Built for raw scale - the Doubao family reportedly handles over a hundred trillion token calls a day, mostly inside China's consumer apps. Hard to access from the West."
  },
  {
   "id": "ernie-5-1",
   "name": "ERNIE 5.1",
   "vendor": "Baidu",
   "family": null,
   "status": "public",
   "access": "Baidu cloud & apps (closed weights)",
   "tags": [
    "paid",
    "api",
    "info-only"
   ],
   "labels": [
    "multilingual",
    "enterprise"
   ],
   "score": {
    "aa": 38,
    "est": true,
    "cat": {
     "coding": 64,
     "reasoning": 64,
     "writing": 60,
     "agents": 58
    }
   },
   "review": "Baidu's flagship stays closed and China-centric - competitive domestically, rarely benchmarked on Western boards."
  },
  {
   "id": "mimo-v2-5-pro",
   "name": "MiMo-V2.5-Pro",
   "vendor": "Xiaomi",
   "family": null,
   "status": "public",
   "access": "Xiaomi cloud & devices (closed)",
   "tags": [
    "paid",
    "info-only"
   ],
   "labels": [
    "vision",
    "speed"
   ],
   "score": {
    "aa": 42.2,
    "est": false,
    "cat": {
     "coding": 70,
     "reasoning": 70,
     "writing": 60,
     "agents": 66
    }
   },
   "review": "Xiaomi's surprisingly strong in-house line (an Omni sibling handles voice/vision on-device) - a sign of how deep the Chinese frontier bench now runs."
  },
  {
   "id": "ling-2-6-flash",
   "name": "Ling 2.6 Flash",
   "vendor": "Ant Group",
   "family": "llama",
   "status": "public",
   "access": "Open weights",
   "tags": [
    "free",
    "open-weights",
    "prompt-target"
   ],
   "labels": [
    "speed",
    "value"
   ],
   "score": {
    "aa": 14.1,
    "est": false,
    "cat": {
     "coding": 32,
     "reasoning": 28,
     "writing": 28,
     "agents": 24
    }
   },
   "review": "Ant Group's efficient open line - built for cheap, fast serving rather than peak intelligence."
  },
  {
   "id": "step-3-7-flash",
   "name": "Step 3.7 Flash",
   "vendor": "StepFun",
   "family": "llama",
   "status": "public",
   "access": "Open weights · API",
   "tags": [
    "free",
    "open-weights",
    "api",
    "prompt-target"
   ],
   "labels": [
    "speed",
    "multilingual"
   ],
   "score": {
    "aa": 30.3,
    "est": false,
    "cat": {
     "coding": 54,
     "reasoning": 52,
     "writing": 50,
     "agents": 46
    }
   },
   "review": "StepFun's fast open tier - part of the broad Chinese open-weight wave now taking ~45% of OpenRouter traffic."
  },
  {
   "id": "minimax-m3",
   "name": "MiniMax M3",
   "vendor": "MiniMax",
   "family": "llama",
   "status": "public",
   "access": "Open weights · API hosts",
   "tags": [
    "free",
    "open-weights",
    "api",
    "prompt-target"
   ],
   "labels": [
    "coding",
    "long-context"
   ],
   "score": {
    "aa": 44.4,
    "est": false,
    "cat": {
     "coding": 82,
     "reasoning": 74,
     "writing": 64,
     "agents": 76
    }
   },
   "review": "MiniMax's new flagship - #2 open model on the index behind GLM-5.2, with a reputation for huge context handling."
  },
  {
   "id": "minimax-m2-7",
   "name": "MiniMax M2.7",
   "vendor": "MiniMax",
   "family": "llama",
   "status": "public",
   "access": "Open weights · API hosts",
   "tags": [
    "free",
    "open-weights",
    "api",
    "prompt-target"
   ],
   "labels": [
    "coding"
   ],
   "score": {
    "aa": 38.1,
    "est": false,
    "cat": {
     "coding": 72,
     "reasoning": 64,
     "writing": 58,
     "agents": 66
    }
   },
   "review": "Strong open coding contender in head-to-head arenas - less known in the West, worth watching."
  },
  {
   "id": "inkling",
   "name": "Inkling",
   "vendor": "Thinking Machines Lab",
   "family": "llama",
   "status": "public",
   "access": "Open weights (Apache 2.0) on Hugging Face · Tinker API (256K ctx)",
   "tags": [
    "free",
    "open-weights",
    "api",
    "prompt-target"
   ],
   "labels": [
    "vision",
    "local",
    "long-context"
   ],
   "spec": {
    "released": "2026-07-15",
    "ctx": "1M",
    "modal": "Text, image, audio"
   },
   "score": {
    "aa": 40.7,
    "est": false,
    "cat": {
     "coding": 62,
     "reasoning": 68,
     "writing": 60,
     "agents": 64
    }
   },
   "review": "Thinking Machines' first open release: 975B-parameter MoE (41B active) reasoning natively over text, images and audio, with controllable thinking effort - the new leading US open-weights base for customization."
  },
  {
   "id": "nemotron-3-ultra",
   "name": "Nemotron 3 Ultra",
   "vendor": "NVIDIA",
   "family": "nemotron",
   "status": "public",
   "spec": {
    "ctx": "1M",
    "speed": "400+ tok/s on some hosts"
   },
   "access": "FREE on OpenRouter (nvidia/nemotron-3-ultra-550b-a55b:free - WhichAI default runner) · open weights · Perplexity Pro",
   "tags": [
    "free",
    "open-weights",
    "api",
    "prompt-target",
    "auto-run"
   ],
   "labels": [
    "speed",
    "value",
    "reasoning"
   ],
   "score": {
    "aa": 37.8,
    "est": false,
    "cat": {
     "coding": 68,
     "reasoning": 68,
     "writing": 58,
     "agents": 62
    }
   },
   "review": "Strongest US open model and extremely fast (400+ tokens/second on some hosts) - and genuinely free via OpenRouter."
  },
  {
   "id": "nemotron-3-super",
   "name": "Nemotron 3 Super",
   "vendor": "NVIDIA",
   "family": "nemotron",
   "status": "public",
   "access": "FREE on OpenRouter (nvidia/nemotron-3-super-120b-a12b:free) · open weights",
   "tags": [
    "free",
    "open-weights",
    "api",
    "prompt-target",
    "auto-run"
   ],
   "labels": [
    "speed",
    "value"
   ],
   "score": {
    "aa": 25.4,
    "est": false,
    "cat": {
     "coding": 48,
     "reasoning": 46,
     "writing": 42,
     "agents": 40
    }
   },
   "review": "The 120B sibling - lighter, faster, still free; good for volume tasks."
  },
  {
   "id": "mistral-medium-3-5",
   "name": "Mistral Medium 3.5",
   "vendor": "Mistral AI",
   "family": "llama",
   "status": "public",
   "access": "Open weights · Le Chat · API",
   "tags": [
    "free",
    "open-weights",
    "api",
    "prompt-target"
   ],
   "labels": [
    "multilingual",
    "value"
   ],
   "score": {
    "aa": 29.9,
    "est": false,
    "cat": {
     "coding": 56,
     "reasoning": 54,
     "writing": 54,
     "agents": 48
    }
   },
   "review": "Mistral's current best on the index - a newer generation that outscores the bigger, older Large 3. Europe's leading open lab."
  },
  {
   "id": "mistral-large-3",
   "name": "Mistral Large 3",
   "vendor": "Mistral AI",
   "family": "llama",
   "status": "public",
   "access": "Open weights · Le Chat free tier · API",
   "tags": [
    "free",
    "open-weights",
    "api",
    "prompt-target"
   ],
   "labels": [
    "multilingual"
   ],
   "score": {
    "aa": 15.9,
    "est": false,
    "cat": {
     "coding": 36,
     "reasoning": 32,
     "writing": 38,
     "agents": 28
    }
   },
   "review": "The 675B MoE that was Europe's flagship in 2025 - the index now places it well behind newer, smaller Mistral releases."
  },
  {
   "id": "mistral-small-4",
   "name": "Mistral Small 4",
   "vendor": "Mistral AI",
   "family": "llama",
   "status": "public",
   "access": "Open weights (Apache 2.0) · laptop-class",
   "tags": [
    "free",
    "open-weights",
    "prompt-target"
   ],
   "labels": [
    "local",
    "value"
   ],
   "score": {
    "aa": 19.6,
    "est": false,
    "cat": {
     "coding": 40,
     "reasoning": 36,
     "writing": 40,
     "agents": 30
    }
   },
   "review": "Compact European open model - a common pick for private, on-premise deployments."
  },
  {
   "id": "ministral-3-14b",
   "name": "Ministral 3 14B",
   "vendor": "Mistral AI",
   "family": "llama",
   "status": "public",
   "access": "Open weights · edge/on-device",
   "tags": [
    "free",
    "open-weights"
   ],
   "labels": [
    "local",
    "speed"
   ],
   "score": {
    "aa": 11.1,
    "est": false,
    "cat": {
     "coding": 26,
     "reasoning": 22,
     "writing": 26,
     "agents": 18
    }
   },
   "review": "Mistral's edge line - built to run on phones and small boxes, not to win benchmarks."
  },
  {
   "id": "mercury-2",
   "name": "Mercury 2",
   "vendor": "Inception Labs",
   "family": null,
   "status": "public",
   "access": "API",
   "tags": [
    "paid",
    "api",
    "info-only"
   ],
   "labels": [
    "speed"
   ],
   "score": {
    "aa": 21.4,
    "est": false,
    "cat": {
     "coding": 52,
     "reasoning": 44,
     "writing": 42,
     "agents": 40
    }
   },
   "review": "A diffusion language model - generates text in parallel rather than token-by-token, hitting extreme speeds. The most interesting architectural outsider on the board."
  },
  {
   "id": "command-a-plus",
   "name": "Command A+",
   "vendor": "Cohere",
   "family": null,
   "status": "public",
   "access": "Enterprise API · open weights for research · private deployments",
   "tags": [
    "paid",
    "api",
    "open-weights",
    "info-only"
   ],
   "labels": [
    "enterprise",
    "multilingual"
   ],
   "score": {
    "aa": 22.5,
    "est": false,
    "cat": {
     "coding": 42,
     "reasoning": 40,
     "writing": 44,
     "agents": 40
    }
   },
   "review": "Enterprise-focused: retrieval-augmented answers, data residency, private deployment - chosen for compliance more than raw benchmarks."
  },
  {
   "id": "nova-2",
   "name": "Nova 2",
   "vendor": "Amazon",
   "family": null,
   "status": "public",
   "access": "AWS Bedrock (enterprise)",
   "tags": [
    "paid",
    "api",
    "info-only"
   ],
   "labels": [
    "enterprise",
    "value"
   ],
   "score": {
    "aa": 22,
    "est": true,
    "cat": {
     "coding": 44,
     "reasoning": 42,
     "writing": 42,
     "agents": 42
    }
   },
   "review": "Amazon's in-house line, priced aggressively inside Bedrock - the default for AWS-native companies, not a benchmark leader."
  },
  {
   "id": "granite-4",
   "name": "Granite 4.0",
   "vendor": "IBM",
   "family": null,
   "status": "public",
   "access": "Open weights (Apache 2.0) · watsonx",
   "tags": [
    "free",
    "open-weights",
    "info-only"
   ],
   "labels": [
    "enterprise",
    "local"
   ],
   "score": {
    "aa": 8,
    "est": true,
    "cat": {
     "coding": 22,
     "reasoning": 18,
     "writing": 20,
     "agents": 16
    }
   },
   "review": "IBM's open enterprise family (hybrid Mamba architecture, tiny-to-mid sizes) - built for governed, on-premise business use, not leaderboards."
  },
  {
   "id": "jamba-large",
   "name": "Jamba Large",
   "vendor": "AI21 Labs",
   "family": null,
   "status": "public",
   "access": "Open weights · API",
   "tags": [
    "free",
    "open-weights",
    "api",
    "info-only"
   ],
   "labels": [
    "long-context",
    "enterprise"
   ],
   "score": {
    "aa": 12,
    "est": true,
    "cat": {
     "coding": 26,
     "reasoning": 24,
     "writing": 30,
     "agents": 20
    }
   },
   "review": "AI21's hybrid SSM-Transformer line - memory-efficient very-long-context processing for enterprises, a different trade-off than pure Transformers."
  },
  {
   "id": "k-exaone",
   "name": "K-Exaone",
   "vendor": "LG AI Research",
   "family": null,
   "status": "public",
   "access": "Korean cloud ecosystem · some open releases",
   "tags": [
    "paid",
    "info-only"
   ],
   "labels": [
    "multilingual",
    "enterprise"
   ],
   "score": {
    "aa": 22.1,
    "est": false,
    "cat": {
     "coding": 46,
     "reasoning": 44,
     "writing": 42,
     "agents": 38
    }
   },
   "review": "Korea's leading domestic line - strong on Korean-language benchmarks, part of the sovereign-AI wave."
  },
  {
   "id": "solar-pro-2",
   "name": "Solar Pro 2",
   "vendor": "Upstage",
   "family": null,
   "status": "public",
   "access": "API · Korean ecosystem",
   "tags": [
    "paid",
    "api",
    "info-only"
   ],
   "labels": [
    "multilingual",
    "value"
   ],
   "score": {
    "aa": 7.8,
    "est": false,
    "cat": {
     "coding": 20,
     "reasoning": 16,
     "writing": 20,
     "agents": 14
    }
   },
   "review": "Upstage's efficient Korean-market model - document AI and enterprise Korean text are its home turf."
  },
  {
   "id": "trinity-large",
   "name": "Trinity Large",
   "vendor": "Arcee AI",
   "family": "llama",
   "status": "public",
   "access": "Open weights",
   "tags": [
    "free",
    "open-weights",
    "prompt-target"
   ],
   "labels": [
    "local",
    "value"
   ],
   "score": {
    "aa": 18.2,
    "est": false,
    "cat": {
     "coding": 46,
     "reasoning": 44,
     "writing": 42,
     "agents": 40
    }
   },
   "review": "A rare US startup training capable open models from scratch - the 'Trinity' line is the credible American answer to the Chinese open wave."
  },
  {
   "id": "sarvam-105b",
   "name": "Sarvam 105B",
   "vendor": "Sarvam AI",
   "family": null,
   "status": "public",
   "access": "Open weights · Indian cloud",
   "tags": [
    "free",
    "open-weights",
    "info-only"
   ],
   "labels": [
    "multilingual"
   ],
   "score": {
    "aa": 11.9,
    "est": false,
    "cat": {
     "coding": 24,
     "reasoning": 22,
     "writing": 26,
     "agents": 18
    }
   },
   "review": "India's sovereign-AI flagship - optimized for Indic languages, a milestone for AI outside the US-China duopoly."
  },
  {
   "id": "falcon-h1",
   "name": "Falcon H1",
   "vendor": "TII (UAE)",
   "family": null,
   "status": "public",
   "access": "Open weights",
   "tags": [
    "free",
    "open-weights",
    "info-only"
   ],
   "labels": [
    "local"
   ],
   "score": {
    "aa": 7,
    "est": true,
    "cat": {
     "coding": 18,
     "reasoning": 16,
     "writing": 18,
     "agents": 12
    }
   },
   "review": "The UAE's hybrid-architecture open family - a persistent, well-funded outsider rather than a frontier contender."
  },
  {
   "id": "reka-flash-3",
   "name": "Reka Flash 3",
   "vendor": "Reka AI",
   "family": null,
   "status": "public",
   "access": "Open weights",
   "tags": [
    "free",
    "open-weights",
    "info-only"
   ],
   "labels": [
    "local",
    "vision"
   ],
   "score": {
    "aa": 8,
    "est": true,
    "cat": {
     "coding": 20,
     "reasoning": 18,
     "writing": 20,
     "agents": 14
    }
   },
   "review": "Small multimodal open model from an ex-DeepMind team - efficient vision-language work on modest hardware."
  },
  {
   "id": "olmo-2",
   "name": "OLMo 2",
   "vendor": "Ai2",
   "family": null,
   "status": "public",
   "access": "Fully open: weights, data and training code",
   "tags": [
    "free",
    "open-weights",
    "info-only"
   ],
   "labels": [
    "local"
   ],
   "score": {
    "aa": 5,
    "est": true,
    "cat": {
     "coding": 14,
     "reasoning": 12,
     "writing": 16,
     "agents": 8
    }
   },
   "review": "The most transparent model line in existence - everything is published, including training data. Built for science, not benchmarks."
  },
  {
   "id": "lfm-2-5",
   "name": "LFM2.5",
   "vendor": "Liquid AI",
   "family": null,
   "status": "public",
   "access": "Open weights · edge/on-device",
   "tags": [
    "free",
    "open-weights",
    "info-only"
   ],
   "labels": [
    "local",
    "speed"
   ],
   "score": {
    "aa": 8.3,
    "est": false,
    "cat": {
     "coding": 18,
     "reasoning": 16,
     "writing": 18,
     "agents": 12
    }
   },
   "review": "'Liquid' non-Transformer architecture squeezing real capability into phone-sized models - watch this space for on-device AI."
  }
 ]
}