{"schemaVersion":"3.0-draft","sources":[{"name":"OpenRouter","role":"availability catalog","url":"https://openrouter.ai/api/v1/models?output_modalities=all"},{"name":"XiaomiMiMo","role":"official model release metadata","url":"https://github.com/XiaomiMiMo/MiMo-V2-Flash"},{"name":"LiveBench","role":"language benchmark history (11 releases)","url":"https://api.github.com/repos/LiveBench/new-livebench/contents/public"},{"name":"Artificial Analysis","role":"language, image, and video benchmark scores","url":"https://artificialanalysis.ai/models","measuredAt":"2026-08-27"},{"name":"MTEB","role":"embedding retrieval benchmark","url":"https://mteb-leaderboard-backend.hf.space/v1/benchmarks/MTEB%28eng%2C%20v2%29/scores","measuredAt":"2026-08-27"},{"name":"LMArena","role":"human-preference text leaderboard","url":"https://datasets-server.huggingface.co/rows?dataset=lmarena-ai%2Fleaderboard-dataset&config=text&split=latest","measuredAt":"2026-08-27"},{"name":"Hugging Face OpenEvals","role":"official multi-benchmark aggregate","url":"https://datasets-server.huggingface.co/rows?dataset=OpenEvals%2Fleaderboard-data&config=default&split=train","measuredAt":"2026-08-27"},{"name":"Hugging Face model eval results","role":"registered model-card and benchmark results","url":"https://huggingface.co/docs/hub/eval-results","measuredAt":"2026-08-27"},{"name":"VibeLeaderboard Intel","role":"related editorial context","url":"/intel"}],"scope":"Canonical releases observed in the source catalogs; deployment routes are collapsed and this is not an index of every community checkpoint or quantization.","capturedAt":"2026-08-27T21:26:33.034Z","modelCount":471,"routeCount":550,"labCount":69,"modalityCounts":{"multimodal":163,"language":157,"image":50,"embedding":33,"audio":41,"video":27},"benchmarkCoverage":{"modelsWithScores":246,"note":"Scores remain source-specific and include a date, scope, and direct source URL. Unmatched releases are left blank rather than fuzzily assigned."},"labs":[{"id":"anthropic","name":"Anthropic","domain":"anthropic.com","major":true,"modelCount":14,"series":[{"id":"claude-opus","label":"Opus","models":[{"id":"anthropic/claude-opus-4","name":"Claude Opus 4","releaseLabel":"4","generationId":"4","generationLabel":"4","variantLabel":"Base","createdAt":"2025-05-22T16:27:25.000Z","kind":"multimodal","contextLength":200000,"inputModalities":["image","text","file"],"outputModalities":["text"],"promptPerMillion":15,"completionPerMillion":75,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Claude Opus 4 is benchmarked as the world’s best coding model, at time of release, bringing sustained performance on complex, long-running tasks and agent workflows. It sets new benchmarks in...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"anthropic/claude-opus-4","url":"https://openrouter.ai/anthropic/claude-opus-4","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"65a873d4-29fc-457c-b403-ede9bad6821b","title":"On Working with Wizards","href":"/intel/65a873d4-29fc-457c-b403-ede9bad6821b","hook":"It gives practitioners a sharp mental model for working with frontier models like GPT-5 Pro and Claude Opus — recognizing when you're 'summoning a wizard' wh…"},{"id":"74485487-cb17-48cf-9db1-262c1a298542","title":"Claude Constitution","href":"/intel/74485487-cb17-48cf-9db1-262c1a298542","hook":"The foundational statement of Claude's values and intended behavior. Short, and worth reading to understand the priorities baked into the model you are promp…"},{"id":"e9fdb80b-b463-4766-99ed-c84d437714d0","title":"Anthropic's Boris Cherny: Why Coding Is Solved, and What Comes Next","href":"/intel/e9fdb80b-b463-4766-99ed-c84d437714d0","hook":"Cherny's claim isn't that AI helps you code faster—it's that the bottleneck has already moved off your keyboard entirely, toward dispatching and reviewing ag…"}]},{"id":"anthropic/claude-opus-4-1","name":"Claude Opus 4.1","releaseLabel":"4.1","generationId":"4.1","generationLabel":"4.1","variantLabel":"Base","createdAt":"2025-08-05T16:33:11.000Z","kind":"multimodal","contextLength":200000,"inputModalities":["image","text","file"],"outputModalities":["text"],"promptPerMillion":15,"completionPerMillion":75,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Claude Opus 4.1 is an updated version of Anthropic’s flagship model, offering improved performance in coding, reasoning, and agentic tasks. It achieves 74.5% on SWE-bench Verified and shows notable gains...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"anthropic/claude-opus-4.1","url":"https://openrouter.ai/anthropic/claude-opus-4.1","free":false,"batch":false},{"provider":"OpenRouter","modelId":"anthropic/claude-opus-4.1:batch","url":"https://openrouter.ai/anthropic/claude-opus-4.1:batch","free":false,"batch":true}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"65a873d4-29fc-457c-b403-ede9bad6821b","title":"On Working with Wizards","href":"/intel/65a873d4-29fc-457c-b403-ede9bad6821b","hook":"It gives practitioners a sharp mental model for working with frontier models like GPT-5 Pro and Claude Opus — recognizing when you're 'summoning a wizard' wh…"},{"id":"74485487-cb17-48cf-9db1-262c1a298542","title":"Claude Constitution","href":"/intel/74485487-cb17-48cf-9db1-262c1a298542","hook":"The foundational statement of Claude's values and intended behavior. Short, and worth reading to understand the priorities baked into the model you are promp…"},{"id":"e9fdb80b-b463-4766-99ed-c84d437714d0","title":"Anthropic's Boris Cherny: Why Coding Is Solved, and What Comes Next","href":"/intel/e9fdb80b-b463-4766-99ed-c84d437714d0","hook":"Cherny's claim isn't that AI helps you code faster—it's that the bottleneck has already moved off your keyboard entirely, toward dispatching and reviewing ag…"}]},{"id":"anthropic/claude-opus-4-5","name":"Claude Opus 4.5","releaseLabel":"4.5","generationId":"4.5","generationLabel":"4.5","variantLabel":"Base","createdAt":"2025-11-24T18:56:20.000Z","kind":"multimodal","contextLength":200000,"inputModalities":["file","image","text"],"outputModalities":["text"],"promptPerMillion":5,"completionPerMillion":25,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"anthropic/claude-opus-4.5","url":"https://openrouter.ai/anthropic/claude-opus-4.5","free":false,"batch":false},{"provider":"OpenRouter","modelId":"anthropic/claude-opus-4.5:batch","url":"https://openrouter.ai/anthropic/claude-opus-4.5:batch","free":false,"batch":true}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"65a873d4-29fc-457c-b403-ede9bad6821b","title":"On Working with Wizards","href":"/intel/65a873d4-29fc-457c-b403-ede9bad6821b","hook":"It gives practitioners a sharp mental model for working with frontier models like GPT-5 Pro and Claude Opus — recognizing when you're 'summoning a wizard' wh…"},{"id":"74485487-cb17-48cf-9db1-262c1a298542","title":"Claude Constitution","href":"/intel/74485487-cb17-48cf-9db1-262c1a298542","hook":"The foundational statement of Claude's values and intended behavior. Short, and worth reading to understand the priorities baked into the model you are promp…"},{"id":"e9fdb80b-b463-4766-99ed-c84d437714d0","title":"Anthropic's Boris Cherny: Why Coding Is Solved, and What Comes Next","href":"/intel/e9fdb80b-b463-4766-99ed-c84d437714d0","hook":"Cherny's claim isn't that AI helps you code faster—it's that the bottleneck has already moved off your keyboard entirely, toward dispatching and reviewing ag…"}]},{"id":"anthropic/claude-opus-4-6","name":"Claude Opus 4.6","releaseLabel":"4.6","generationId":"4.6","generationLabel":"4.6","variantLabel":"Base","createdAt":"2026-02-04T15:30:50.000Z","kind":"multimodal","contextLength":1000000,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":5,"completionPerMillion":25,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"anthropic/claude-opus-4.6","url":"https://openrouter.ai/anthropic/claude-opus-4.6","free":false,"batch":false},{"provider":"OpenRouter","modelId":"anthropic/claude-opus-4.6:batch","url":"https://openrouter.ai/anthropic/claude-opus-4.6:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"LiveBench overall","score":75.1,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1503,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 3 · 72,359 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"65a873d4-29fc-457c-b403-ede9bad6821b","title":"On Working with Wizards","href":"/intel/65a873d4-29fc-457c-b403-ede9bad6821b","hook":"It gives practitioners a sharp mental model for working with frontier models like GPT-5 Pro and Claude Opus — recognizing when you're 'summoning a wizard' wh…"},{"id":"74485487-cb17-48cf-9db1-262c1a298542","title":"Claude Constitution","href":"/intel/74485487-cb17-48cf-9db1-262c1a298542","hook":"The foundational statement of Claude's values and intended behavior. Short, and worth reading to understand the priorities baked into the model you are promp…"},{"id":"e9fdb80b-b463-4766-99ed-c84d437714d0","title":"Anthropic's Boris Cherny: Why Coding Is Solved, and What Comes Next","href":"/intel/e9fdb80b-b463-4766-99ed-c84d437714d0","hook":"Cherny's claim isn't that AI helps you code faster—it's that the bottleneck has already moved off your keyboard entirely, toward dispatching and reviewing ag…"}]},{"id":"anthropic/claude-opus-4-7","name":"Claude Opus 4.7","releaseLabel":"4.7","generationId":"4.7","generationLabel":"4.7","variantLabel":"Base","createdAt":"2026-04-16T14:51:40.000Z","kind":"multimodal","contextLength":1000000,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":5,"completionPerMillion":25,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"anthropic/claude-opus-4.7","url":"https://openrouter.ai/anthropic/claude-opus-4.7","free":false,"batch":false},{"provider":"OpenRouter","modelId":"anthropic/claude-opus-4.7-fast","url":"https://openrouter.ai/anthropic/claude-opus-4.7-fast","free":false,"batch":false},{"provider":"OpenRouter","modelId":"anthropic/claude-opus-4.7:batch","url":"https://openrouter.ai/anthropic/claude-opus-4.7:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"LiveBench overall","score":77,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1490,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 7 · 60,203 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"65a873d4-29fc-457c-b403-ede9bad6821b","title":"On Working with Wizards","href":"/intel/65a873d4-29fc-457c-b403-ede9bad6821b","hook":"It gives practitioners a sharp mental model for working with frontier models like GPT-5 Pro and Claude Opus — recognizing when you're 'summoning a wizard' wh…"},{"id":"74485487-cb17-48cf-9db1-262c1a298542","title":"Claude Constitution","href":"/intel/74485487-cb17-48cf-9db1-262c1a298542","hook":"The foundational statement of Claude's values and intended behavior. Short, and worth reading to understand the priorities baked into the model you are promp…"},{"id":"e9fdb80b-b463-4766-99ed-c84d437714d0","title":"Anthropic's Boris Cherny: Why Coding Is Solved, and What Comes Next","href":"/intel/e9fdb80b-b463-4766-99ed-c84d437714d0","hook":"Cherny's claim isn't that AI helps you code faster—it's that the bottleneck has already moved off your keyboard entirely, toward dispatching and reviewing ag…"}]},{"id":"anthropic/claude-opus-4-8","name":"Claude Opus 4.8","releaseLabel":"4.8","generationId":"4.8","generationLabel":"4.8","variantLabel":"Base","createdAt":"2026-05-27T18:04:51.000Z","kind":"multimodal","contextLength":1000000,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":5,"completionPerMillion":25,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"anthropic/claude-opus-4.8","url":"https://openrouter.ai/anthropic/claude-opus-4.8","free":false,"batch":false},{"provider":"OpenRouter","modelId":"anthropic/claude-opus-4.8-fast","url":"https://openrouter.ai/anthropic/claude-opus-4.8-fast","free":false,"batch":false},{"provider":"OpenRouter","modelId":"anthropic/claude-opus-4.8:batch","url":"https://openrouter.ai/anthropic/claude-opus-4.8:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"LiveBench overall","score":77.6,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1461,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 30 · 44,676 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"65a873d4-29fc-457c-b403-ede9bad6821b","title":"On Working with Wizards","href":"/intel/65a873d4-29fc-457c-b403-ede9bad6821b","hook":"It gives practitioners a sharp mental model for working with frontier models like GPT-5 Pro and Claude Opus — recognizing when you're 'summoning a wizard' wh…"},{"id":"74485487-cb17-48cf-9db1-262c1a298542","title":"Claude Constitution","href":"/intel/74485487-cb17-48cf-9db1-262c1a298542","hook":"The foundational statement of Claude's values and intended behavior. Short, and worth reading to understand the priorities baked into the model you are promp…"},{"id":"e9fdb80b-b463-4766-99ed-c84d437714d0","title":"Anthropic's Boris Cherny: Why Coding Is Solved, and What Comes Next","href":"/intel/e9fdb80b-b463-4766-99ed-c84d437714d0","hook":"Cherny's claim isn't that AI helps you code faster—it's that the bottleneck has already moved off your keyboard entirely, toward dispatching and reviewing ag…"}]},{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","releaseLabel":"5","generationId":"5","generationLabel":"5","variantLabel":"Base","createdAt":"2026-07-24T17:02:24.000Z","kind":"multimodal","contextLength":1000000,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":5,"completionPerMillion":25,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"anthropic/claude-opus-5","url":"https://openrouter.ai/anthropic/claude-opus-5","free":false,"batch":false},{"provider":"OpenRouter","modelId":"anthropic/claude-opus-5-fast","url":"https://openrouter.ai/anthropic/claude-opus-5-fast","free":false,"batch":false},{"provider":"OpenRouter","modelId":"anthropic/claude-opus-5:batch","url":"https://openrouter.ai/anthropic/claude-opus-5:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"AA Intelligence Index","score":63.1,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-opus-5","measuredAt":"2026-08-27","scope":"Current public index; highest tested effort configuration"},{"benchmark":"AA Agentic Index","score":59.2,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-opus-5","measuredAt":"2026-08-27","scope":"Current public agentic composite"},{"benchmark":"SciCode","score":55.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-opus-5","measuredAt":"2026-08-27","scope":"Scientific coding evaluation"},{"benchmark":"AA Omniscience Index","score":37.1,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-opus-5","measuredAt":"2026-08-27","scope":"Current public knowledge and hallucination index"},{"benchmark":"MLCR overall","score":55.6,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-opus-5","measuredAt":"2026-08-27","scope":"Machine-learning code reasoning evaluation"},{"benchmark":"GDPval-AA normalized","score":66.1,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-opus-5","measuredAt":"2026-08-27","scope":"Normalized real-world knowledge-work evaluation"},{"benchmark":"τ²-bench Banking","score":42.1,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-opus-5","measuredAt":"2026-08-27","scope":"Tool-agent benchmark in the banking domain"},{"benchmark":"Terminal-Bench 2.1","score":89.1,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-opus-5","measuredAt":"2026-08-27","scope":"Terminal-agent task completion"},{"benchmark":"LCR","score":75.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-opus-5","measuredAt":"2026-08-27","scope":"Long-context reasoning evaluation"},{"benchmark":"Humanity's Last Exam","score":54.9,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-opus-5","measuredAt":"2026-08-27","scope":"Broad expert-level knowledge evaluation"},{"benchmark":"GPQA","score":93.2,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-opus-5","measuredAt":"2026-08-27","scope":"Graduate-level science questions"},{"benchmark":"CritPt","score":29.1,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-opus-5","measuredAt":"2026-08-27","scope":"Critical-point reasoning evaluation"},{"benchmark":"MMMU Pro","score":84.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-opus-5","measuredAt":"2026-08-27","scope":"Multimodal expert reasoning evaluation"},{"benchmark":"Analyst Agent","score":53.8,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-opus-5","measuredAt":"2026-08-27","scope":"Analyst workflow agent tasks"},{"benchmark":"Enterprise Ops Gym","score":47.5,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-opus-5","measuredAt":"2026-08-27","scope":"Enterprise operations agent tasks"},{"benchmark":"LMArena overall","score":1504,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 1 · 27,610 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"65a873d4-29fc-457c-b403-ede9bad6821b","title":"On Working with Wizards","href":"/intel/65a873d4-29fc-457c-b403-ede9bad6821b","hook":"It gives practitioners a sharp mental model for working with frontier models like GPT-5 Pro and Claude Opus — recognizing when you're 'summoning a wizard' wh…"},{"id":"74485487-cb17-48cf-9db1-262c1a298542","title":"Claude Constitution","href":"/intel/74485487-cb17-48cf-9db1-262c1a298542","hook":"The foundational statement of Claude's values and intended behavior. Short, and worth reading to understand the priorities baked into the model you are promp…"},{"id":"e9fdb80b-b463-4766-99ed-c84d437714d0","title":"Anthropic's Boris Cherny: Why Coding Is Solved, and What Comes Next","href":"/intel/e9fdb80b-b463-4766-99ed-c84d437714d0","hook":"Cherny's claim isn't that AI helps you code faster—it's that the bottleneck has already moved off your keyboard entirely, toward dispatching and reviewing ag…"}]}]},{"id":"claude-sonnet","label":"Sonnet","models":[{"id":"anthropic/claude-sonnet-4","name":"Claude Sonnet 4","releaseLabel":"4","generationId":"4","generationLabel":"4","variantLabel":"Base","createdAt":"2025-05-22T16:12:51.000Z","kind":"multimodal","contextLength":1000000,"inputModalities":["image","text","file"],"outputModalities":["text"],"promptPerMillion":3,"completionPerMillion":15,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Claude Sonnet 4 significantly enhances the capabilities of its predecessor, Sonnet 3.7, excelling in both coding and reasoning tasks with improved precision and controllability. Achieving state-of-the-art performance on SWE-bench (72.7%),...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"anthropic/claude-sonnet-4","url":"https://openrouter.ai/anthropic/claude-sonnet-4","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"c1f469e7-8a5e-4722-bf30-d8dc83c310db","title":"Rhys's MCP Retrospective","href":"/intel/c1f469e7-8a5e-4722-bf30-d8dc83c310db","hook":"Rhys Sullivan's analysis of why MCP didn't take off when it launched alongside models like Sonnet 3.5 and GPT-4o, and what comes next for agent-to-tool conne…"},{"id":"825de045-40a5-4cfc-98bb-b12dc6763f21","title":"Real AI Agents and Real Work","href":"/intel/825de045-40a5-4cfc-98bb-b12dc6763f21","hook":"A grounded read on where AI agents actually produce economic value versus noise, anchored by a concrete demo of Claude Sonnet 4.5 reproducing an academic pap…"},{"id":"74485487-cb17-48cf-9db1-262c1a298542","title":"Claude Constitution","href":"/intel/74485487-cb17-48cf-9db1-262c1a298542","hook":"The foundational statement of Claude's values and intended behavior. Short, and worth reading to understand the priorities baked into the model you are promp…"}]},{"id":"anthropic/claude-sonnet-4-5","name":"Claude Sonnet 4.5","releaseLabel":"4.5","generationId":"4.5","generationLabel":"4.5","variantLabel":"Base","createdAt":"2025-09-29T16:01:16.000Z","kind":"multimodal","contextLength":1000000,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":3,"completionPerMillion":15,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"anthropic/claude-sonnet-4.5","url":"https://openrouter.ai/anthropic/claude-sonnet-4.5","free":false,"batch":false},{"provider":"OpenRouter","modelId":"anthropic/claude-sonnet-4.5:batch","url":"https://openrouter.ai/anthropic/claude-sonnet-4.5:batch","free":false,"batch":true}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"c1f469e7-8a5e-4722-bf30-d8dc83c310db","title":"Rhys's MCP Retrospective","href":"/intel/c1f469e7-8a5e-4722-bf30-d8dc83c310db","hook":"Rhys Sullivan's analysis of why MCP didn't take off when it launched alongside models like Sonnet 3.5 and GPT-4o, and what comes next for agent-to-tool conne…"},{"id":"825de045-40a5-4cfc-98bb-b12dc6763f21","title":"Real AI Agents and Real Work","href":"/intel/825de045-40a5-4cfc-98bb-b12dc6763f21","hook":"A grounded read on where AI agents actually produce economic value versus noise, anchored by a concrete demo of Claude Sonnet 4.5 reproducing an academic pap…"},{"id":"74485487-cb17-48cf-9db1-262c1a298542","title":"Claude Constitution","href":"/intel/74485487-cb17-48cf-9db1-262c1a298542","hook":"The foundational statement of Claude's values and intended behavior. Short, and worth reading to understand the priorities baked into the model you are promp…"}]},{"id":"anthropic/claude-sonnet-4-6","name":"Claude Sonnet 4.6","releaseLabel":"4.6","generationId":"4.6","generationLabel":"4.6","variantLabel":"Base","createdAt":"2026-02-17T15:43:10.000Z","kind":"multimodal","contextLength":1000000,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":3,"completionPerMillion":15,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"anthropic/claude-sonnet-4.6","url":"https://openrouter.ai/anthropic/claude-sonnet-4.6","free":false,"batch":false},{"provider":"OpenRouter","modelId":"anthropic/claude-sonnet-4.6:batch","url":"https://openrouter.ai/anthropic/claude-sonnet-4.6:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"LiveBench overall","score":73.4,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1458,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 31 · 66,512 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"c1f469e7-8a5e-4722-bf30-d8dc83c310db","title":"Rhys's MCP Retrospective","href":"/intel/c1f469e7-8a5e-4722-bf30-d8dc83c310db","hook":"Rhys Sullivan's analysis of why MCP didn't take off when it launched alongside models like Sonnet 3.5 and GPT-4o, and what comes next for agent-to-tool conne…"},{"id":"825de045-40a5-4cfc-98bb-b12dc6763f21","title":"Real AI Agents and Real Work","href":"/intel/825de045-40a5-4cfc-98bb-b12dc6763f21","hook":"A grounded read on where AI agents actually produce economic value versus noise, anchored by a concrete demo of Claude Sonnet 4.5 reproducing an academic pap…"},{"id":"74485487-cb17-48cf-9db1-262c1a298542","title":"Claude Constitution","href":"/intel/74485487-cb17-48cf-9db1-262c1a298542","hook":"The foundational statement of Claude's values and intended behavior. Short, and worth reading to understand the priorities baked into the model you are promp…"}]},{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","releaseLabel":"5","generationId":"5","generationLabel":"5","variantLabel":"Base","createdAt":"2026-06-30T18:11:23.000Z","kind":"multimodal","contextLength":1000000,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":2,"completionPerMillion":10,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"anthropic/claude-sonnet-5","url":"https://openrouter.ai/anthropic/claude-sonnet-5","free":false,"batch":false},{"provider":"OpenRouter","modelId":"anthropic/claude-sonnet-5:batch","url":"https://openrouter.ai/anthropic/claude-sonnet-5:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"LiveBench overall","score":76.6,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1442,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 57 · 27,366 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"c1f469e7-8a5e-4722-bf30-d8dc83c310db","title":"Rhys's MCP Retrospective","href":"/intel/c1f469e7-8a5e-4722-bf30-d8dc83c310db","hook":"Rhys Sullivan's analysis of why MCP didn't take off when it launched alongside models like Sonnet 3.5 and GPT-4o, and what comes next for agent-to-tool conne…"},{"id":"825de045-40a5-4cfc-98bb-b12dc6763f21","title":"Real AI Agents and Real Work","href":"/intel/825de045-40a5-4cfc-98bb-b12dc6763f21","hook":"A grounded read on where AI agents actually produce economic value versus noise, anchored by a concrete demo of Claude Sonnet 4.5 reproducing an academic pap…"},{"id":"74485487-cb17-48cf-9db1-262c1a298542","title":"Claude Constitution","href":"/intel/74485487-cb17-48cf-9db1-262c1a298542","hook":"The foundational statement of Claude's values and intended behavior. Short, and worth reading to understand the priorities baked into the model you are promp…"}]}]},{"id":"claude-haiku","label":"Haiku","models":[{"id":"anthropic/claude-3-haiku","name":"Claude 3 Haiku","releaseLabel":"3","generationId":"3","generationLabel":"3","variantLabel":"Haiku","createdAt":"2024-03-13T00:00:00.000Z","kind":"multimodal","contextLength":200000,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0.25,"completionPerMillion":1.25,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Claude 3 Haiku is Anthropic's fastest and most compact model for near-instant responsiveness. Quick and accurate targeted performance. See the launch announcement and benchmark results [here](https://www.anthropic.com/news/claude-3-haiku) #multimodal","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"anthropic/claude-3-haiku","url":"https://openrouter.ai/anthropic/claude-3-haiku","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"74485487-cb17-48cf-9db1-262c1a298542","title":"Claude Constitution","href":"/intel/74485487-cb17-48cf-9db1-262c1a298542","hook":"The foundational statement of Claude's values and intended behavior. Short, and worth reading to understand the priorities baked into the model you are promp…"},{"id":"e9fdb80b-b463-4766-99ed-c84d437714d0","title":"Anthropic's Boris Cherny: Why Coding Is Solved, and What Comes Next","href":"/intel/e9fdb80b-b463-4766-99ed-c84d437714d0","hook":"Cherny's claim isn't that AI helps you code faster—it's that the bottleneck has already moved off your keyboard entirely, toward dispatching and reviewing ag…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]},{"id":"anthropic/claude-haiku-4-5","name":"Claude Haiku 4.5","releaseLabel":"4.5","generationId":"4.5","generationLabel":"4.5","variantLabel":"Base","createdAt":"2025-10-15T17:00:38.000Z","kind":"multimodal","contextLength":200000,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":1,"completionPerMillion":5,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"anthropic/claude-haiku-4.5","url":"https://openrouter.ai/anthropic/claude-haiku-4.5","free":false,"batch":false},{"provider":"OpenRouter","modelId":"anthropic/claude-haiku-4.5:batch","url":"https://openrouter.ai/anthropic/claude-haiku-4.5:batch","free":false,"batch":true}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"74485487-cb17-48cf-9db1-262c1a298542","title":"Claude Constitution","href":"/intel/74485487-cb17-48cf-9db1-262c1a298542","hook":"The foundational statement of Claude's values and intended behavior. Short, and worth reading to understand the priorities baked into the model you are promp…"},{"id":"e9fdb80b-b463-4766-99ed-c84d437714d0","title":"Anthropic's Boris Cherny: Why Coding Is Solved, and What Comes Next","href":"/intel/e9fdb80b-b463-4766-99ed-c84d437714d0","hook":"Cherny's claim isn't that AI helps you code faster—it's that the bottleneck has already moved off your keyboard entirely, toward dispatching and reviewing ag…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]}]},{"id":"claude-fable","label":"Fable","models":[{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","releaseLabel":"5","generationId":"5","generationLabel":"5","variantLabel":"Base","createdAt":"2026-06-09T12:18:35.000Z","kind":"multimodal","contextLength":1000000,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":10,"completionPerMillion":50,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"anthropic/claude-fable-5","url":"https://openrouter.ai/anthropic/claude-fable-5","free":false,"batch":false},{"provider":"OpenRouter","modelId":"anthropic/claude-fable-5:batch","url":"https://openrouter.ai/anthropic/claude-fable-5:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"AA Intelligence Index","score":62.1,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-fable-5","measuredAt":"2026-08-27","scope":"Current public index; highest tested effort configuration"},{"benchmark":"AA Agentic Index","score":56.6,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-fable-5","measuredAt":"2026-08-27","scope":"Current public agentic composite"},{"benchmark":"SciCode","score":60.2,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-fable-5","measuredAt":"2026-08-27","scope":"Scientific coding evaluation"},{"benchmark":"AA Omniscience Index","score":43.3,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-fable-5","measuredAt":"2026-08-27","scope":"Current public knowledge and hallucination index"},{"benchmark":"MLCR overall","score":64.4,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-fable-5","measuredAt":"2026-08-27","scope":"Machine-learning code reasoning evaluation"},{"benchmark":"GDPval-AA normalized","score":61.1,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-fable-5","measuredAt":"2026-08-27","scope":"Normalized real-world knowledge-work evaluation"},{"benchmark":"τ²-bench","score":98.5,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-fable-5","measuredAt":"2026-08-27","scope":"Tool-agent benchmark across supported domains"},{"benchmark":"τ²-bench Banking","score":38.1,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-fable-5","measuredAt":"2026-08-27","scope":"Tool-agent benchmark in the banking domain"},{"benchmark":"Terminal-Bench Hard","score":62.9,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-fable-5","measuredAt":"2026-08-27","scope":"Hard terminal-agent tasks"},{"benchmark":"Terminal-Bench 2.1","score":84.6,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-fable-5","measuredAt":"2026-08-27","scope":"Terminal-agent task completion"},{"benchmark":"LCR","score":76.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-fable-5","measuredAt":"2026-08-27","scope":"Long-context reasoning evaluation"},{"benchmark":"IFBench","score":63.5,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-fable-5","measuredAt":"2026-08-27","scope":"Instruction-following evaluation"},{"benchmark":"Humanity's Last Exam","score":55.5,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-fable-5","measuredAt":"2026-08-27","scope":"Broad expert-level knowledge evaluation"},{"benchmark":"GPQA","score":92.6,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-fable-5","measuredAt":"2026-08-27","scope":"Graduate-level science questions"},{"benchmark":"CritPt","score":28.6,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-fable-5","measuredAt":"2026-08-27","scope":"Critical-point reasoning evaluation"},{"benchmark":"Analyst Agent","score":48.8,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-fable-5","measuredAt":"2026-08-27","scope":"Analyst workflow agent tasks"},{"benchmark":"Enterprise Ops Gym","score":51.1,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-fable-5","measuredAt":"2026-08-27","scope":"Enterprise operations agent tasks"},{"benchmark":"LMArena overall","score":1495,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 5 · 24,331 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"6c4bdfa6-ae35-4ce5-9f72-f60b74d7a68f","title":"A Field Guide to Fable: Finding Your Unknowns","href":"/intel/6c4bdfa6-ae35-4ce5-9f72-f60b74d7a68f","hook":"If your AI coding sessions keep drifting off-track, the problem usually isn't the model — it's unresolved unknowns you didn't know you had. This framework gi…"},{"id":"42e6243a-941d-4982-b370-1a8aefa4046c","title":"Claude Fable 5 Analysis (Interconnects)","href":"/intel/42e6243a-941d-4982-b370-1a8aefa4046c","hook":"If you build on frontier models, you need to know when the vendor silently alters model behavior for certain query classes — this piece documents exactly tha…"},{"id":"74485487-cb17-48cf-9db1-262c1a298542","title":"Claude Constitution","href":"/intel/74485487-cb17-48cf-9db1-262c1a298542","hook":"The foundational statement of Claude's values and intended behavior. Short, and worth reading to understand the priorities baked into the model you are promp…"}]}]}]},{"id":"openai","name":"OpenAI","domain":"openai.com","major":true,"modelCount":68,"series":[{"id":"gpt","label":"GPT","models":[{"id":"openai/gpt-3.5-turbo","name":"GPT-3.5 Turbo","releaseLabel":"3.5 Turbo","generationId":"3.5","generationLabel":"3.5","variantLabel":"Turbo","createdAt":"2023-05-28T00:00:00.000Z","kind":"language","contextLength":16385,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.5,"completionPerMillion":1.5,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks. Training data up to Sep 2021.","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"openai/gpt-3.5-turbo","url":"https://openrouter.ai/openai/gpt-3.5-turbo","free":false,"batch":false},{"provider":"OpenRouter","modelId":"openai/gpt-3.5-turbo:batch","url":"https://openrouter.ai/openai/gpt-3.5-turbo:batch","free":false,"batch":true}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-4","name":"GPT-4","releaseLabel":"4","generationId":"4","generationLabel":"4","variantLabel":"Base","createdAt":"2023-05-28T00:00:00.000Z","kind":"language","contextLength":8191,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":30,"completionPerMillion":60,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"OpenAI's flagship model, GPT-4 is a large-scale multimodal language model capable of solving difficult problems with greater accuracy than previous models due to its broader general knowledge and advanced reasoning...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"openai/gpt-4","url":"https://openrouter.ai/openai/gpt-4","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-3.5-turbo-16k","name":"GPT-3.5 Turbo 16k","releaseLabel":"3.5 Turbo 16k","generationId":"3.5","generationLabel":"3.5","variantLabel":"Turbo 16k","createdAt":"2023-08-28T00:00:00.000Z","kind":"language","contextLength":16385,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":3,"completionPerMillion":4,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"This model offers four times the context length of gpt-3.5-turbo, allowing it to support approximately 20 pages of text in a single request at a higher cost. Training data: up...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"openai/gpt-3.5-turbo-16k","url":"https://openrouter.ai/openai/gpt-3.5-turbo-16k","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-3.5-turbo-instruct","name":"GPT-3.5 Turbo Instruct","releaseLabel":"3.5 Turbo Instruct","generationId":"3.5","generationLabel":"3.5","variantLabel":"Turbo Instruct","createdAt":"2023-09-28T00:00:00.000Z","kind":"language","contextLength":4095,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":1.5,"completionPerMillion":2,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"This model is a variant of GPT-3.5 Turbo tuned for instructional prompts and omitting chat-related optimizations. Training data: up to Sep 2021.","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"openai/gpt-3.5-turbo-instruct","url":"https://openrouter.ai/openai/gpt-3.5-turbo-instruct","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-3.5-turbo-0613","name":"GPT-3.5 Turbo (older v0613)","releaseLabel":"3.5 Turbo (older v0613)","generationId":"3.5","generationLabel":"3.5","variantLabel":"Turbo 0613","createdAt":"2024-01-25T00:00:00.000Z","kind":"language","contextLength":4095,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":1,"completionPerMillion":2,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks. Training data up to Sep 2021.","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"openai/gpt-3.5-turbo-0613","url":"https://openrouter.ai/openai/gpt-3.5-turbo-0613","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-4-turbo-preview","name":"GPT-4 Turbo Preview","releaseLabel":"4 Turbo Preview","generationId":"4","generationLabel":"4","variantLabel":"Turbo Preview","createdAt":"2024-01-25T00:00:00.000Z","kind":"language","contextLength":128000,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":10,"completionPerMillion":30,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"The preview GPT-4 model with improved instruction following, JSON mode, reproducible outputs, parallel function calling, and more. Training data: up to Dec 2023. **Note:** heavily rate limited by OpenAI while...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"openai/gpt-4-turbo-preview","url":"https://openrouter.ai/openai/gpt-4-turbo-preview","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-4-turbo","name":"GPT-4 Turbo","releaseLabel":"4 Turbo","generationId":"4","generationLabel":"4","variantLabel":"Turbo","createdAt":"2024-04-09T00:00:00.000Z","kind":"multimodal","contextLength":128000,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":10,"completionPerMillion":30,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling. Training data: up to December 2023.","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"openai/gpt-4-turbo","url":"https://openrouter.ai/openai/gpt-4-turbo","free":false,"batch":false},{"provider":"OpenRouter","modelId":"openai/gpt-4-turbo:batch","url":"https://openrouter.ai/openai/gpt-4-turbo:batch","free":false,"batch":true}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-4o","name":"GPT-4o","releaseLabel":"4o","generationId":"4o","generationLabel":"4o","variantLabel":"Base","createdAt":"2024-05-13T00:00:00.000Z","kind":"multimodal","contextLength":128000,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":2.5,"completionPerMillion":10,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"openai/gpt-4o","url":"https://openrouter.ai/openai/gpt-4o","free":false,"batch":false},{"provider":"OpenRouter","modelId":"openai/gpt-4o:batch","url":"https://openrouter.ai/openai/gpt-4o:batch","free":false,"batch":true}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-4o-2024-05-13","name":"GPT-4o (2024-05-13)","releaseLabel":"4o (2024-05-13)","generationId":"4o","generationLabel":"4o","variantLabel":"(2024 05 13) · 2024-05-13","createdAt":"2024-05-13T00:00:00.000Z","kind":"multimodal","contextLength":128000,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":5,"completionPerMillion":15,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"openai/gpt-4o-2024-05-13","url":"https://openrouter.ai/openai/gpt-4o-2024-05-13","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":55.2,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2024_08_31.csv","measuredAt":"2024-08-31","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1301,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 226 · 112,881 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-4o-mini","name":"GPT-4o-mini","releaseLabel":"4o-mini","generationId":"4o","generationLabel":"4o","variantLabel":"Mini","createdAt":"2024-07-18T00:00:00.000Z","kind":"multimodal","contextLength":128000,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":0.15,"completionPerMillion":0.6,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"openai/gpt-4o-mini","url":"https://openrouter.ai/openai/gpt-4o-mini","free":false,"batch":false},{"provider":"OpenRouter","modelId":"openai/gpt-4o-mini:batch","url":"https://openrouter.ai/openai/gpt-4o-mini:batch","free":false,"batch":true}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-4o-mini-2024-07-18","name":"GPT-4o-mini (2024-07-18)","releaseLabel":"4o-mini (2024-07-18)","generationId":"4o","generationLabel":"4o","variantLabel":"Mini · 2024-07-18","createdAt":"2024-07-18T00:00:00.000Z","kind":"multimodal","contextLength":128000,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":0.15,"completionPerMillion":0.6,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"openai/gpt-4o-mini-2024-07-18","url":"https://openrouter.ai/openai/gpt-4o-mini-2024-07-18","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":42.8,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2025_04_25.csv","measuredAt":"2025-04-25","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1287,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 241 · 68,715 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-4o-2024-08-06","name":"GPT-4o (2024-08-06)","releaseLabel":"4o (2024-08-06)","generationId":"4o","generationLabel":"4o","variantLabel":"(2024 08 06) · 2024-08-06","createdAt":"2024-08-06T00:00:00.000Z","kind":"multimodal","contextLength":128000,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":2.5,"completionPerMillion":10,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"The 2024-08-06 version of GPT-4o offers improved performance in structured outputs, with the ability to supply a JSON schema in the respone_format. Read more [here](https://openai.com/index/introducing-structured-outputs-in-the-api/). GPT-4o (\"o\" for \"omni\") is...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"openai/gpt-4o-2024-08-06","url":"https://openrouter.ai/openai/gpt-4o-2024-08-06","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":51.3,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2025_04_02.csv","measuredAt":"2025-04-02","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1283,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 245 · 45,499 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-4o-2024-11-20","name":"GPT-4o (2024-11-20)","releaseLabel":"4o (2024-11-20)","generationId":"4o","generationLabel":"4o","variantLabel":"(2024 11 20) · 2024-11-20","createdAt":"2024-11-20T18:33:14.000Z","kind":"multimodal","contextLength":128000,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":2.5,"completionPerMillion":10,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"The 2024-11-20 version of GPT-4o offers a leveled-up creative writing ability with more natural, engaging, and tailored writing to improve relevance & readability. It’s also better at working with uploaded...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-4o-2024-11-20","url":"https://openrouter.ai/openai/gpt-4o-2024-11-20","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":53.1,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2025_04_25.csv","measuredAt":"2025-04-25","scope":"Mean across that public release's task columns"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-4.1-nano-2025-04-14","name":"GPT-4.1 Nano","releaseLabel":"4.1 Nano","generationId":"4.1","generationLabel":"4.1","variantLabel":"Nano","createdAt":"2025-04-14T17:22:49.000Z","kind":"multimodal","contextLength":1047576,"inputModalities":["image","text","file"],"outputModalities":["text"],"promptPerMillion":0.09999999999999999,"completionPerMillion":0.39999999999999997,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-4.1-nano","url":"https://openrouter.ai/openai/gpt-4.1-nano","free":false,"batch":false},{"provider":"OpenRouter","modelId":"openai/gpt-4.1-nano:batch","url":"https://openrouter.ai/openai/gpt-4.1-nano:batch","free":false,"batch":true}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-4.1-mini-2025-04-14","name":"GPT-4.1 Mini","releaseLabel":"4.1 Mini","generationId":"4.1","generationLabel":"4.1","variantLabel":"Mini","createdAt":"2025-04-14T17:23:01.000Z","kind":"multimodal","contextLength":1047576,"inputModalities":["image","text","file"],"outputModalities":["text"],"promptPerMillion":0.39999999999999997,"completionPerMillion":1.5999999999999999,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-4.1-mini","url":"https://openrouter.ai/openai/gpt-4.1-mini","free":false,"batch":false},{"provider":"OpenRouter","modelId":"openai/gpt-4.1-mini:batch","url":"https://openrouter.ai/openai/gpt-4.1-mini:batch","free":false,"batch":true}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-4.1-2025-04-14","name":"GPT-4.1","releaseLabel":"4.1","generationId":"4.1","generationLabel":"4.1","variantLabel":"Base","createdAt":"2025-04-14T17:23:05.000Z","kind":"multimodal","contextLength":1047576,"inputModalities":["image","text","file"],"outputModalities":["text"],"promptPerMillion":2,"completionPerMillion":8,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-4.1","url":"https://openrouter.ai/openai/gpt-4.1","free":false,"batch":false},{"provider":"OpenRouter","modelId":"openai/gpt-4.1:batch","url":"https://openrouter.ai/openai/gpt-4.1:batch","free":false,"batch":true}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-5-nano-2025-08-07","name":"GPT-5 Nano","releaseLabel":"5 Nano","generationId":"5","generationLabel":"5","variantLabel":"Nano","createdAt":"2025-08-07T17:23:22.000Z","kind":"multimodal","contextLength":400000,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":0.049999999999999996,"completionPerMillion":0.39999999999999997,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-5-nano","url":"https://openrouter.ai/openai/gpt-5-nano","free":false,"batch":false},{"provider":"OpenRouter","modelId":"openai/gpt-5-nano:batch","url":"https://openrouter.ai/openai/gpt-5-nano:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"LiveBench overall","score":33.7,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1320,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 212 · 8,292 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-5-mini-2025-08-07","name":"GPT-5 Mini","releaseLabel":"5 Mini","generationId":"5","generationLabel":"5","variantLabel":"Mini","createdAt":"2025-08-07T17:23:27.000Z","kind":"multimodal","contextLength":400000,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":0.25,"completionPerMillion":2,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-5-mini","url":"https://openrouter.ai/openai/gpt-5-mini","free":false,"batch":false},{"provider":"OpenRouter","modelId":"openai/gpt-5-mini:batch","url":"https://openrouter.ai/openai/gpt-5-mini:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"LiveBench overall","score":52.4,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1373,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 161 · 27,140 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-5-2025-08-07","name":"GPT-5","releaseLabel":"5","generationId":"5","generationLabel":"5","variantLabel":"Base","createdAt":"2025-08-07T17:23:33.000Z","kind":"multimodal","contextLength":400000,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":1.25,"completionPerMillion":10,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-5","url":"https://openrouter.ai/openai/gpt-5","free":false,"batch":false},{"provider":"OpenRouter","modelId":"openai/gpt-5:batch","url":"https://openrouter.ai/openai/gpt-5:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"LiveBench overall","score":75.6,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2025_05_30.csv","measuredAt":"2025-05-30","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1406,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 131 · 32,081 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-5-pro-2025-10-06","name":"GPT-5 Pro","releaseLabel":"5 Pro","generationId":"5","generationLabel":"5","variantLabel":"Pro","createdAt":"2025-10-06T18:51:03.000Z","kind":"multimodal","contextLength":400000,"inputModalities":["image","text","file"],"outputModalities":["text"],"promptPerMillion":15,"completionPerMillion":120,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GPT-5 Pro is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-5-pro","url":"https://openrouter.ai/openai/gpt-5-pro","free":false,"batch":false},{"provider":"OpenRouter","modelId":"openai/gpt-5-pro:batch","url":"https://openrouter.ai/openai/gpt-5-pro:batch","free":false,"batch":true}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-5.1-20251113","name":"GPT-5.1","releaseLabel":"5.1","generationId":"5.1","generationLabel":"5.1","variantLabel":"Base","createdAt":"2025-11-13T18:58:25.000Z","kind":"multimodal","contextLength":400000,"inputModalities":["image","text","file"],"outputModalities":["text"],"promptPerMillion":1.25,"completionPerMillion":10,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-5.1","url":"https://openrouter.ai/openai/gpt-5.1","free":false,"batch":false},{"provider":"OpenRouter","modelId":"openai/gpt-5.1:batch","url":"https://openrouter.ai/openai/gpt-5.1:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"LMArena overall","score":1441,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 59 · 40,945 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-5.2-20251211","name":"GPT-5.2","releaseLabel":"5.2","generationId":"5.2","generationLabel":"5.2","variantLabel":"Base","createdAt":"2025-12-10T18:02:55.000Z","kind":"multimodal","contextLength":400000,"inputModalities":["file","image","text"],"outputModalities":["text"],"promptPerMillion":1.75,"completionPerMillion":14,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-5.2","url":"https://openrouter.ai/openai/gpt-5.2","free":false,"batch":false},{"provider":"OpenRouter","modelId":"openai/gpt-5.2:batch","url":"https://openrouter.ai/openai/gpt-5.2:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"LMArena overall","score":1416,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 114 · 47,920 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-5.2-pro-20251211","name":"GPT-5.2 Pro","releaseLabel":"5.2 Pro","generationId":"5.2","generationLabel":"5.2","variantLabel":"Pro","createdAt":"2025-12-10T18:03:00.000Z","kind":"multimodal","contextLength":400000,"inputModalities":["image","text","file"],"outputModalities":["text"],"promptPerMillion":21,"completionPerMillion":168,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GPT-5.2 Pro is OpenAI’s most advanced model, offering major improvements in agentic coding and long context performance over GPT-5 Pro. It is optimized for complex tasks that require step-by-step reasoning,...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-5.2-pro","url":"https://openrouter.ai/openai/gpt-5.2-pro","free":false,"batch":false},{"provider":"OpenRouter","modelId":"openai/gpt-5.2-pro:batch","url":"https://openrouter.ai/openai/gpt-5.2-pro:batch","free":false,"batch":true}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-5.2-chat-20251211","name":"GPT-5.2 Chat","releaseLabel":"5.2 Chat","generationId":"5.2","generationLabel":"5.2","variantLabel":"Chat","createdAt":"2025-12-10T18:03:03.000Z","kind":"multimodal","contextLength":128000,"inputModalities":["file","image","text"],"outputModalities":["text"],"promptPerMillion":1.75,"completionPerMillion":14,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"GPT-5.2 Chat (AKA Instant) is the fast, lightweight member of the 5.2 family, optimized for low-latency chat while retaining strong general intelligence. It uses adaptive reasoning to selectively “think” on...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-5.2-chat","url":"https://openrouter.ai/openai/gpt-5.2-chat","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1439,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 66 · 34,339 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-5.4-20260305","name":"GPT-5.4","releaseLabel":"5.4","generationId":"5.4","generationLabel":"5.4","variantLabel":"Base","createdAt":"2026-03-05T18:12:32.000Z","kind":"multimodal","contextLength":1050000,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":2.5,"completionPerMillion":15,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-5.4","url":"https://openrouter.ai/openai/gpt-5.4","free":false,"batch":false},{"provider":"OpenRouter","modelId":"openai/gpt-5.4:batch","url":"https://openrouter.ai/openai/gpt-5.4:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"LiveBench overall","score":78.8,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1470,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 23 · 60,708 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-5.4-pro-20260305","name":"GPT-5.4 Pro","releaseLabel":"5.4 Pro","generationId":"5.4","generationLabel":"5.4","variantLabel":"Pro","createdAt":"2026-03-05T18:12:46.000Z","kind":"multimodal","contextLength":1050000,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":30,"completionPerMillion":180,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-5.4-pro","url":"https://openrouter.ai/openai/gpt-5.4-pro","free":false,"batch":false},{"provider":"OpenRouter","modelId":"openai/gpt-5.4-pro:batch","url":"https://openrouter.ai/openai/gpt-5.4-pro:batch","free":false,"batch":true}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-5.4-mini-20260317","name":"GPT-5.4 Mini","releaseLabel":"5.4 Mini","generationId":"5.4","generationLabel":"5.4","variantLabel":"Mini","createdAt":"2026-03-17T11:49:38.000Z","kind":"multimodal","contextLength":400000,"inputModalities":["file","image","text"],"outputModalities":["text"],"promptPerMillion":0.75,"completionPerMillion":4.5,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-5.4-mini","url":"https://openrouter.ai/openai/gpt-5.4-mini","free":false,"batch":false},{"provider":"OpenRouter","modelId":"openai/gpt-5.4-mini:batch","url":"https://openrouter.ai/openai/gpt-5.4-mini:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"LiveBench overall","score":66.6,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1412,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 120 · 59,502 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-5.4-nano-20260317","name":"GPT-5.4 Nano","releaseLabel":"5.4 Nano","generationId":"5.4","generationLabel":"5.4","variantLabel":"Nano","createdAt":"2026-03-17T11:49:47.000Z","kind":"multimodal","contextLength":400000,"inputModalities":["file","image","text"],"outputModalities":["text"],"promptPerMillion":0.19999999999999998,"completionPerMillion":1.25,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-5.4-nano","url":"https://openrouter.ai/openai/gpt-5.4-nano","free":false,"batch":false},{"provider":"OpenRouter","modelId":"openai/gpt-5.4-nano:batch","url":"https://openrouter.ai/openai/gpt-5.4-nano:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"LiveBench overall","score":70.8,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1373,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 162 · 58,551 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-5.5-20260423","name":"GPT-5.5","releaseLabel":"5.5","generationId":"5.5","generationLabel":"5.5","variantLabel":"Base","createdAt":"2026-04-24T17:31:33.000Z","kind":"multimodal","contextLength":1050000,"inputModalities":["file","image","text"],"outputModalities":["text"],"promptPerMillion":5,"completionPerMillion":30,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-5.5","url":"https://openrouter.ai/openai/gpt-5.5","free":false,"batch":false},{"provider":"OpenRouter","modelId":"openai/gpt-5.5:batch","url":"https://openrouter.ai/openai/gpt-5.5:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"LiveBench overall","score":80.8,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1471,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 22 · 59,545 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-5.5-pro-20260423","name":"GPT-5.5 Pro","releaseLabel":"5.5 Pro","generationId":"5.5","generationLabel":"5.5","variantLabel":"Pro","createdAt":"2026-04-24T17:31:36.000Z","kind":"multimodal","contextLength":1050000,"inputModalities":["file","image","text"],"outputModalities":["text"],"promptPerMillion":30,"completionPerMillion":180,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GPT-5.5 Pro is OpenAI’s high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-5.5-pro","url":"https://openrouter.ai/openai/gpt-5.5-pro","free":false,"batch":false},{"provider":"OpenRouter","modelId":"openai/gpt-5.5-pro:batch","url":"https://openrouter.ai/openai/gpt-5.5-pro:batch","free":false,"batch":true}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-5.6-sol-20260709","name":"GPT-5.6 Sol","releaseLabel":"5.6 Sol","generationId":"5.6","generationLabel":"5.6","variantLabel":"Sol","createdAt":"2026-07-09T09:54:10.000Z","kind":"multimodal","contextLength":1050000,"inputModalities":["file","image","text"],"outputModalities":["text"],"promptPerMillion":2,"completionPerMillion":10,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"openai/gpt-5.6-sol","url":"https://openrouter.ai/openai/gpt-5.6-sol","free":false,"batch":false,"configuration":"Standard"},{"provider":"OpenRouter","modelId":"openai/gpt-5.6-sol-pro","url":"https://openrouter.ai/openai/gpt-5.6-sol-pro","free":false,"batch":false,"configuration":"Pro"},{"provider":"OpenRouter","modelId":"openai/gpt-5.6-sol:batch","url":"https://openrouter.ai/openai/gpt-5.6-sol:batch","free":false,"batch":true,"configuration":"Standard"},{"provider":"OpenRouter","modelId":"openai/gpt-5.6-sol-pro:batch","url":"https://openrouter.ai/openai/gpt-5.6-sol-pro:batch","free":false,"batch":true,"configuration":"Pro"}],"benchmarks":[{"benchmark":"AA Intelligence Index","score":60.9,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-sol","measuredAt":"2026-08-27","scope":"Current public index; highest tested effort configuration"},{"benchmark":"AA Agentic Index","score":57.8,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-sol","measuredAt":"2026-08-27","scope":"Current public agentic composite"},{"benchmark":"SciCode","score":56.1,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-sol","measuredAt":"2026-08-27","scope":"Scientific coding evaluation"},{"benchmark":"AA Omniscience Index","score":22,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-sol","measuredAt":"2026-08-27","scope":"Current public knowledge and hallucination index"},{"benchmark":"MLCR overall","score":26.1,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-sol","measuredAt":"2026-08-27","scope":"Machine-learning code reasoning evaluation"},{"benchmark":"GDPval-AA normalized","score":60.6,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-sol","measuredAt":"2026-08-27","scope":"Normalized real-world knowledge-work evaluation"},{"benchmark":"ITBench SRE","score":56.2,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-sol","measuredAt":"2026-08-27","scope":"Site-reliability engineering agent tasks"},{"benchmark":"τ²-bench","score":85.1,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-sol","measuredAt":"2026-08-27","scope":"Tool-agent benchmark across supported domains"},{"benchmark":"τ²-bench Banking","score":44.3,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-sol","measuredAt":"2026-08-27","scope":"Tool-agent benchmark in the banking domain"},{"benchmark":"Terminal-Bench Hard","score":65.9,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-sol","measuredAt":"2026-08-27","scope":"Hard terminal-agent tasks"},{"benchmark":"Terminal-Bench 2.1","score":88,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-sol","measuredAt":"2026-08-27","scope":"Terminal-agent task completion"},{"benchmark":"LCR","score":77.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-sol","measuredAt":"2026-08-27","scope":"Long-context reasoning evaluation"},{"benchmark":"IFBench","score":72.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-sol","measuredAt":"2026-08-27","scope":"Instruction-following evaluation"},{"benchmark":"Humanity's Last Exam","score":49.5,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-sol","measuredAt":"2026-08-27","scope":"Broad expert-level knowledge evaluation"},{"benchmark":"GPQA","score":94.1,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-sol","measuredAt":"2026-08-27","scope":"Graduate-level science questions"},{"benchmark":"CritPt","score":32.3,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-sol","measuredAt":"2026-08-27","scope":"Critical-point reasoning evaluation"},{"benchmark":"MMMU Pro","score":83.4,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-sol","measuredAt":"2026-08-27","scope":"Multimodal expert reasoning evaluation"},{"benchmark":"Analyst Agent","score":47.5,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-sol","measuredAt":"2026-08-27","scope":"Analyst workflow agent tasks"},{"benchmark":"Enterprise Ops Gym","score":42.9,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-sol","measuredAt":"2026-08-27","scope":"Enterprise operations agent tasks"},{"benchmark":"LMArena overall","score":1454,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 35 · 19,541 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-5.6-terra-20260709","name":"GPT-5.6 Terra","releaseLabel":"5.6 Terra","generationId":"5.6","generationLabel":"5.6","variantLabel":"Terra","createdAt":"2026-07-09T09:54:17.000Z","kind":"multimodal","contextLength":1050000,"inputModalities":["file","image","text"],"outputModalities":["text"],"promptPerMillion":2,"completionPerMillion":12,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"openai/gpt-5.6-terra","url":"https://openrouter.ai/openai/gpt-5.6-terra","free":false,"batch":false,"configuration":"Standard"},{"provider":"OpenRouter","modelId":"openai/gpt-5.6-terra-pro","url":"https://openrouter.ai/openai/gpt-5.6-terra-pro","free":false,"batch":false,"configuration":"Pro"},{"provider":"OpenRouter","modelId":"openai/gpt-5.6-terra:batch","url":"https://openrouter.ai/openai/gpt-5.6-terra:batch","free":false,"batch":true,"configuration":"Standard"},{"provider":"OpenRouter","modelId":"openai/gpt-5.6-terra-pro:batch","url":"https://openrouter.ai/openai/gpt-5.6-terra-pro:batch","free":false,"batch":true,"configuration":"Pro"}],"benchmarks":[{"benchmark":"AA Intelligence Index","score":56.6,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-terra","measuredAt":"2026-08-27","scope":"Current public index; highest tested effort configuration"},{"benchmark":"AA Agentic Index","score":50.2,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-terra","measuredAt":"2026-08-27","scope":"Current public agentic composite"},{"benchmark":"SciCode","score":53.9,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-terra","measuredAt":"2026-08-27","scope":"Scientific coding evaluation"},{"benchmark":"AA Omniscience Index","score":0.1,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-terra","measuredAt":"2026-08-27","scope":"Current public knowledge and hallucination index"},{"benchmark":"MLCR overall","score":31.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-terra","measuredAt":"2026-08-27","scope":"Machine-learning code reasoning evaluation"},{"benchmark":"GDPval-AA normalized","score":53.3,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-terra","measuredAt":"2026-08-27","scope":"Normalized real-world knowledge-work evaluation"},{"benchmark":"ITBench SRE","score":51,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-terra","measuredAt":"2026-08-27","scope":"Site-reliability engineering agent tasks"},{"benchmark":"τ²-bench","score":86.3,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-terra","measuredAt":"2026-08-27","scope":"Tool-agent benchmark across supported domains"},{"benchmark":"τ²-bench Banking","score":40.2,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-terra","measuredAt":"2026-08-27","scope":"Tool-agent benchmark in the banking domain"},{"benchmark":"Terminal-Bench Hard","score":57.6,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-terra","measuredAt":"2026-08-27","scope":"Hard terminal-agent tasks"},{"benchmark":"Terminal-Bench 2.1","score":88,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-terra","measuredAt":"2026-08-27","scope":"Terminal-agent task completion"},{"benchmark":"LCR","score":79.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-terra","measuredAt":"2026-08-27","scope":"Long-context reasoning evaluation"},{"benchmark":"IFBench","score":71.2,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-terra","measuredAt":"2026-08-27","scope":"Instruction-following evaluation"},{"benchmark":"Humanity's Last Exam","score":42.9,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-terra","measuredAt":"2026-08-27","scope":"Broad expert-level knowledge evaluation"},{"benchmark":"GPQA","score":92.5,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-terra","measuredAt":"2026-08-27","scope":"Graduate-level science questions"},{"benchmark":"CritPt","score":30,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-terra","measuredAt":"2026-08-27","scope":"Critical-point reasoning evaluation"},{"benchmark":"APEX Agents","score":38.9,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-terra","measuredAt":"2026-08-27","scope":"Professional agent task evaluation"},{"benchmark":"MMMU Pro","score":80.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-terra","measuredAt":"2026-08-27","scope":"Multimodal expert reasoning evaluation"},{"benchmark":"Enterprise Ops Gym","score":38.5,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-terra","measuredAt":"2026-08-27","scope":"Enterprise operations agent tasks"},{"benchmark":"LMArena overall","score":1446,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 47 · 20,216 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]},{"id":"openai/gpt-5.6-luna-20260709","name":"GPT-5.6 Luna","releaseLabel":"5.6 Luna","generationId":"5.6","generationLabel":"5.6","variantLabel":"Luna","createdAt":"2026-07-09T09:54:24.000Z","kind":"multimodal","contextLength":1050000,"inputModalities":["file","image","text"],"outputModalities":["text"],"promptPerMillion":0.19999999999999998,"completionPerMillion":1.2,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"openai/gpt-5.6-luna","url":"https://openrouter.ai/openai/gpt-5.6-luna","free":false,"batch":false,"configuration":"Standard"},{"provider":"OpenRouter","modelId":"openai/gpt-5.6-luna-pro","url":"https://openrouter.ai/openai/gpt-5.6-luna-pro","free":false,"batch":false,"configuration":"Pro"},{"provider":"OpenRouter","modelId":"openai/gpt-5.6-luna:batch","url":"https://openrouter.ai/openai/gpt-5.6-luna:batch","free":false,"batch":true,"configuration":"Standard"},{"provider":"OpenRouter","modelId":"openai/gpt-5.6-luna-pro:batch","url":"https://openrouter.ai/openai/gpt-5.6-luna-pro:batch","free":false,"batch":true,"configuration":"Pro"}],"benchmarks":[{"benchmark":"AA Intelligence Index","score":52.3,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-luna","measuredAt":"2026-08-27","scope":"Current public index; highest tested effort configuration"},{"benchmark":"AA Agentic Index","score":46.9,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-luna","measuredAt":"2026-08-27","scope":"Current public agentic composite"},{"benchmark":"SciCode","score":52.5,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-luna","measuredAt":"2026-08-27","scope":"Scientific coding evaluation"},{"benchmark":"AA Omniscience Index","score":-10.3,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-luna","measuredAt":"2026-08-27","scope":"Current public knowledge and hallucination index"},{"benchmark":"MLCR overall","score":19.4,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-luna","measuredAt":"2026-08-27","scope":"Machine-learning code reasoning evaluation"},{"benchmark":"GDPval-AA normalized","score":53.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-luna","measuredAt":"2026-08-27","scope":"Normalized real-world knowledge-work evaluation"},{"benchmark":"ITBench SRE","score":40.3,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-luna","measuredAt":"2026-08-27","scope":"Site-reliability engineering agent tasks"},{"benchmark":"τ²-bench Banking","score":31.1,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-luna","measuredAt":"2026-08-27","scope":"Tool-agent benchmark in the banking domain"},{"benchmark":"Terminal-Bench 2.1","score":80.9,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-luna","measuredAt":"2026-08-27","scope":"Terminal-agent task completion"},{"benchmark":"LCR","score":78.3,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-luna","measuredAt":"2026-08-27","scope":"Long-context reasoning evaluation"},{"benchmark":"Humanity's Last Exam","score":39.5,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-luna","measuredAt":"2026-08-27","scope":"Broad expert-level knowledge evaluation"},{"benchmark":"GPQA","score":91.1,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-luna","measuredAt":"2026-08-27","scope":"Graduate-level science questions"},{"benchmark":"CritPt","score":20.6,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-luna","measuredAt":"2026-08-27","scope":"Critical-point reasoning evaluation"},{"benchmark":"APEX Agents","score":35.8,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-luna","measuredAt":"2026-08-27","scope":"Professional agent task evaluation"},{"benchmark":"MMMU Pro","score":78.6,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-luna","measuredAt":"2026-08-27","scope":"Multimodal expert reasoning evaluation"},{"benchmark":"Enterprise Ops Gym","score":40.8,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-luna","measuredAt":"2026-08-27","scope":"Enterprise operations agent tasks"},{"benchmark":"LMArena overall","score":1429,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 83 · 20,693 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"1a4ecd25-b6de-4434-b2ee-6188d842930c","title":"From GPT-2 to gpt-oss: Analyzing the Architectural Advances","href":"/intel/1a4ecd25-b6de-4434-b2ee-6188d842930c","hook":"If you're evaluating or building on OpenAI's gpt-oss models, this walks through their exact architectural choices (MoE, sliding-window attention, MXFP4 quant…"},{"id":"10f1fcac-d669-4114-ba97-41404c1a0e5a","title":"Neural Networks: Zero to Hero","href":"/intel/10f1fcac-d669-4114-ba97-41404c1a0e5a","hook":"Karpathy's full course from micrograd to nanoGPT, built from scratch in Python. The canonical path from \"I use models\" to \"I understand them.\" Long, but noth…"}]}]},{"id":"o-series","label":"o-series","models":[{"id":"openai/o1-2024-12-17","name":"o1","releaseLabel":"o1","generationId":"o1","generationLabel":"O1","variantLabel":"Base","createdAt":"2024-12-17T18:26:39.000Z","kind":"multimodal","contextLength":200000,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":15,"completionPerMillion":60,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"The latest and strongest model family from OpenAI, o1 is designed to spend more time thinking before responding. The o1 model series is trained with large-scale reinforcement learning to reason...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/o1","url":"https://openrouter.ai/openai/o1","free":false,"batch":false},{"provider":"OpenRouter","modelId":"openai/o1:batch","url":"https://openrouter.ai/openai/o1:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"LiveBench overall","score":67.5,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2024_08_31.csv","measuredAt":"2024-08-31","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1353,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 185 · 31,122 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]},{"id":"openai/o3-mini-2025-01-31","name":"o3 Mini","releaseLabel":"o3 Mini","generationId":"o3","generationLabel":"O3","variantLabel":"Mini","createdAt":"2025-01-31T19:28:41.000Z","kind":"language","contextLength":200000,"inputModalities":["text","file"],"outputModalities":["text"],"promptPerMillion":1.1,"completionPerMillion":4.4,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"OpenAI o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and coding. This model supports the `reasoning_effort` parameter, which can be set to...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/o3-mini","url":"https://openrouter.ai/openai/o3-mini","free":false,"batch":false},{"provider":"OpenRouter","modelId":"openai/o3-mini:batch","url":"https://openrouter.ai/openai/o3-mini:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"LMArena overall","score":1337,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 199 · 18,589 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]},{"id":"openai/o3-mini-high-2025-01-31","name":"o3 Mini High","releaseLabel":"o3 Mini High","generationId":"o3","generationLabel":"O3","variantLabel":"Mini High","createdAt":"2025-02-12T15:03:31.000Z","kind":"language","contextLength":200000,"inputModalities":["text","file"],"outputModalities":["text"],"promptPerMillion":1.1,"completionPerMillion":4.4,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"OpenAI o3-mini-high is the same model as [o3-mini](/openai/o3-mini) with reasoning_effort set to high. o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/o3-mini-high","url":"https://openrouter.ai/openai/o3-mini-high","free":false,"batch":false},{"provider":"OpenRouter","modelId":"openai/o3-mini-high:batch","url":"https://openrouter.ai/openai/o3-mini-high:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"LMArena overall","score":1337,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 199 · 18,589 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]},{"id":"openai/o1-pro","name":"o1-pro","releaseLabel":"o1-pro","generationId":"o1","generationLabel":"O1","variantLabel":"Pro","createdAt":"2025-03-19T22:26:51.000Z","kind":"multimodal","contextLength":200000,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":150,"completionPerMillion":600,"imageOutput":null,"supportsTools":false,"supportsReasoning":true,"description":"The o1 series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o1-pro model uses more compute to think harder and provide...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/o1-pro","url":"https://openrouter.ai/openai/o1-pro","free":false,"batch":false},{"provider":"OpenRouter","modelId":"openai/o1-pro:batch","url":"https://openrouter.ai/openai/o1-pro:batch","free":false,"batch":true}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]},{"id":"openai/o4-mini-2025-04-16","name":"o4 Mini","releaseLabel":"o4 Mini","generationId":"o4","generationLabel":"O4","variantLabel":"Mini","createdAt":"2025-04-16T16:29:02.000Z","kind":"multimodal","contextLength":200000,"inputModalities":["image","text","file"],"outputModalities":["text"],"promptPerMillion":1.1,"completionPerMillion":4.4,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining strong multimodal and agentic capabilities. It supports tool use and demonstrates competitive reasoning...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/o4-mini","url":"https://openrouter.ai/openai/o4-mini","free":false,"batch":false},{"provider":"OpenRouter","modelId":"openai/o4-mini:batch","url":"https://openrouter.ai/openai/o4-mini:batch","free":false,"batch":true}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]},{"id":"openai/o3-2025-04-16","name":"o3","releaseLabel":"o3","generationId":"o3","generationLabel":"O3","variantLabel":"Base","createdAt":"2025-04-16T17:10:57.000Z","kind":"multimodal","contextLength":200000,"inputModalities":["image","text","file"],"outputModalities":["text"],"promptPerMillion":2,"completionPerMillion":8,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"o3 is a well-rounded and powerful model across domains. It sets a new standard for math, science, coding, and visual reasoning tasks. It also excels at technical writing and instruction-following....","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/o3","url":"https://openrouter.ai/openai/o3","free":false,"batch":false},{"provider":"OpenRouter","modelId":"openai/o3:batch","url":"https://openrouter.ai/openai/o3:batch","free":false,"batch":true}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]},{"id":"openai/o4-mini-high-2025-04-16","name":"o4 Mini High","releaseLabel":"o4 Mini High","generationId":"o4","generationLabel":"O4","variantLabel":"Mini High","createdAt":"2025-04-16T17:23:32.000Z","kind":"multimodal","contextLength":200000,"inputModalities":["image","text","file"],"outputModalities":["text"],"promptPerMillion":1.1,"completionPerMillion":4.4,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"OpenAI o4-mini-high is the same model as [o4-mini](/openai/o4-mini) with reasoning_effort set to high. OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/o4-mini-high","url":"https://openrouter.ai/openai/o4-mini-high","free":false,"batch":false},{"provider":"OpenRouter","modelId":"openai/o4-mini-high:batch","url":"https://openrouter.ai/openai/o4-mini-high:batch","free":false,"batch":true}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]},{"id":"openai/o3-pro-2025-06-10","name":"o3 Pro","releaseLabel":"o3 Pro","generationId":"o3","generationLabel":"O3","variantLabel":"Pro","createdAt":"2025-06-10T23:32:32.000Z","kind":"multimodal","contextLength":200000,"inputModalities":["text","file","image"],"outputModalities":["text"],"promptPerMillion":20,"completionPerMillion":80,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"The o-series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o3-pro model uses more compute to think harder and provide consistently...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/o3-pro","url":"https://openrouter.ai/openai/o3-pro","free":false,"batch":false},{"provider":"OpenRouter","modelId":"openai/o3-pro:batch","url":"https://openrouter.ai/openai/o3-pro:batch","free":false,"batch":true}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]}]},{"id":"gpt-codex","label":"GPT Codex","models":[{"id":"openai/gpt-5-codex","name":"GPT-5 Codex","releaseLabel":"GPT-5 Codex","generationId":"5","generationLabel":"5","variantLabel":"GPT Codex","createdAt":"2025-09-23T16:03:23.000Z","kind":"multimodal","contextLength":400000,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0.625,"completionPerMillion":5,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GPT-5-Codex is a specialized version of GPT-5 optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-5-codex:batch","url":"https://openrouter.ai/openai/gpt-5-codex:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"LiveBench overall","score":79.6,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2025_05_30.csv","measuredAt":"2025-05-30","scope":"Mean across that public release's task columns"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]},{"id":"openai/gpt-5.1-codex-mini-20251113","name":"GPT-5.1-Codex-Mini","releaseLabel":"GPT-5.1-Codex-Mini","generationId":"5.1","generationLabel":"5.1","variantLabel":"Mini","createdAt":"2025-11-13T18:17:00.000Z","kind":"multimodal","contextLength":400000,"inputModalities":["image","text"],"outputModalities":["text"],"promptPerMillion":0.25,"completionPerMillion":2,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GPT-5.1-Codex-Mini is a smaller and faster version of GPT-5.1-Codex","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-5.1-codex-mini","url":"https://openrouter.ai/openai/gpt-5.1-codex-mini","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":60.8,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]},{"id":"openai/gpt-5.1-codex-20251113","name":"GPT-5.1-Codex","releaseLabel":"GPT-5.1-Codex","generationId":"5.1","generationLabel":"5.1","variantLabel":"GPT Codex","createdAt":"2025-11-13T18:58:18.000Z","kind":"multimodal","contextLength":400000,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":1.25,"completionPerMillion":10,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GPT-5.1-Codex is a specialized version of GPT-5.1 optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-5.1-codex","url":"https://openrouter.ai/openai/gpt-5.1-codex","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":69.3,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]},{"id":"openai/gpt-5.1-codex-max-20251204","name":"GPT-5.1-Codex-Max","releaseLabel":"GPT-5.1-Codex-Max","generationId":"5.1","generationLabel":"5.1","variantLabel":"Max","createdAt":"2025-12-04T20:08:54.000Z","kind":"multimodal","contextLength":400000,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":1.25,"completionPerMillion":10,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GPT-5.1-Codex-Max is OpenAI’s latest agentic coding model, designed for long-running, high-context software development tasks. It is based on an updated version of the 5.1 reasoning stack and trained on agentic...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-5.1-codex-max","url":"https://openrouter.ai/openai/gpt-5.1-codex-max","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":74.4,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]},{"id":"openai/gpt-5.2-codex-20260114","name":"GPT-5.2-Codex","releaseLabel":"GPT-5.2-Codex","generationId":"5.2","generationLabel":"5.2","variantLabel":"GPT Codex","createdAt":"2026-01-14T16:48:35.000Z","kind":"multimodal","contextLength":400000,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":1.75,"completionPerMillion":14,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GPT-5.2-Codex is an upgraded version of GPT-5.1-Codex optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-5.2-codex","url":"https://openrouter.ai/openai/gpt-5.2-codex","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":74,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]},{"id":"openai/gpt-5.3-codex-20260224","name":"GPT-5.3-Codex","releaseLabel":"GPT-5.3-Codex","generationId":"5.3","generationLabel":"5.3","variantLabel":"GPT Codex","createdAt":"2026-02-24T18:52:44.000Z","kind":"multimodal","contextLength":400000,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":1.75,"completionPerMillion":14,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-5.3-codex","url":"https://openrouter.ai/openai/gpt-5.3-codex","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":72,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]}]},{"id":"gpt-image","label":"GPT Image","models":[{"id":"openai/gpt-5-image","name":"GPT-5 Image","releaseLabel":"GPT-5 Image","generationId":"5","generationLabel":"5","variantLabel":"Base","createdAt":"2025-10-14T13:19:46.000Z","kind":"image","contextLength":400000,"inputModalities":["image","text","file"],"outputModalities":["image","text"],"promptPerMillion":10,"completionPerMillion":10,"imageOutput":40,"supportsTools":false,"supportsReasoning":true,"description":"[GPT-5](https://openrouter.ai/openai/gpt-5) Image combines OpenAI's GPT-5 model with state-of-the-art image generation capabilities. It offers major improvements in reasoning, code quality, and user experience while incorporating GPT Image 1's superior instruction following,...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-5-image","url":"https://openrouter.ai/openai/gpt-5-image","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]},{"id":"openai/gpt-5-image-mini","name":"GPT-5 Image Mini","releaseLabel":"GPT-5 Image Mini","generationId":"5","generationLabel":"5","variantLabel":"Mini","createdAt":"2025-10-16T14:23:03.000Z","kind":"image","contextLength":400000,"inputModalities":["file","image","text"],"outputModalities":["image","text"],"promptPerMillion":2.5,"completionPerMillion":2,"imageOutput":8,"supportsTools":false,"supportsReasoning":true,"description":"GPT-5 Image Mini combines OpenAI's advanced language capabilities, powered by [GPT-5 Mini](https://openrouter.ai/openai/gpt-5-mini), with GPT Image 1 Mini for efficient image generation. This natively multimodal model features superior instruction following, text...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-5-image-mini","url":"https://openrouter.ai/openai/gpt-5-image-mini","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]},{"id":"openai/gpt-5.4-image-2-20260421","name":"GPT-5.4 Image 2","releaseLabel":"GPT-5.4 Image 2","generationId":"5.4","generationLabel":"5.4","variantLabel":"2","createdAt":"2026-04-21T18:52:08.000Z","kind":"image","contextLength":272000,"inputModalities":["image","text","file"],"outputModalities":["image","text"],"promptPerMillion":8,"completionPerMillion":15,"imageOutput":30,"supportsTools":false,"supportsReasoning":true,"description":"[GPT-5.4](https://openrouter.ai/openai/gpt-5.4) Image 2 combines OpenAI's GPT-5.4 model with state-of-the-art image generation capabilities from GPT Image 2. It enables rich multimodal workflows, allowing users to seamlessly move between reasoning, coding, and...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-5.4-image-2","url":"https://openrouter.ai/openai/gpt-5.4-image-2","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]},{"id":"openai/gpt-image-1","name":"GPT Image 1","releaseLabel":"1","generationId":"1","generationLabel":"1","variantLabel":"Base","createdAt":"2026-06-24T01:31:53.000Z","kind":"image","contextLength":400000,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":10,"completionPerMillion":10,"imageOutput":40,"supportsTools":false,"supportsReasoning":false,"description":"OpenAI's GPT Image 1 generates and edits images via the dedicated Images API. Features accurate text rendering, transparent backgrounds, and up to 16 reference images for edits.","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"openai/gpt-image-1","url":"https://openrouter.ai/openai/gpt-image-1","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]},{"id":"openai/gpt-image-1-mini","name":"GPT Image 1 Mini","releaseLabel":"1 Mini","generationId":"1","generationLabel":"1","variantLabel":"Mini","createdAt":"2026-06-24T01:31:53.000Z","kind":"image","contextLength":400000,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":2.5,"completionPerMillion":2.5,"imageOutput":8,"supportsTools":false,"supportsReasoning":false,"description":"A cost-efficient variant of GPT Image 1 for high-quality image generation at reduced latency and cost via OpenAI's dedicated Images API.","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"openai/gpt-image-1-mini","url":"https://openrouter.ai/openai/gpt-image-1-mini","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]},{"id":"openai/gpt-image-2","name":"GPT Image 2","releaseLabel":"2","generationId":"2","generationLabel":"2","variantLabel":"Base","createdAt":"2026-06-24T01:31:54.000Z","kind":"image","contextLength":400000,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":8,"completionPerMillion":8,"imageOutput":30,"supportsTools":false,"supportsReasoning":false,"description":"OpenAI's latest image generation model. Supports high-fidelity image generation and editing via the dedicated Images API.","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"openai/gpt-image-2","url":"https://openrouter.ai/openai/gpt-image-2","free":false,"batch":false}],"benchmarks":[{"benchmark":"AA Image Arena Elo","score":1370,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/image/leaderboard/text-to-image","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]}]},{"id":"embeddings","label":"Embeddings","models":[{"id":"openai/text-embedding-3-small","name":"Text Embedding 3 Small","releaseLabel":"Text Embedding 3 Small","generationId":"3","generationLabel":"3","variantLabel":"Small","createdAt":"2025-10-30T20:50:55.000Z","kind":"embedding","contextLength":8192,"inputModalities":["text"],"outputModalities":["embeddings"],"promptPerMillion":0.02,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"text-embedding-3-small is OpenAI's improved, more performant version of the ada embedding model. Embeddings are a numerical representation of text that can be used to measure the relatedness between two pieces...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/text-embedding-3-small","url":"https://openrouter.ai/openai/text-embedding-3-small","free":false,"batch":false}],"benchmarks":[{"benchmark":"MTEB English retrieval","score":53.5,"unit":"%","sourceName":"MTEB","sourceUrl":"https://leaderboard.mteb.org/","measuredAt":"2026-08-27","scope":"Mean retrieval score · leaderboard rank 112"}],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[{"id":"cefdb845-e661-43e5-8120-6ae2ed0fe4e5","title":"Learning Word Embedding","href":"/intel/cefdb845-e661-43e5-8120-6ae2ed0fe4e5","hook":"A clear walkthrough of how free-text words become numeric vectors — from one-hot encoding to learned embeddings — grounding the intuition behind the embeddin…"},{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"}]},{"id":"openai/text-embedding-3-large","name":"Text Embedding 3 Large","releaseLabel":"Text Embedding 3 Large","generationId":"3","generationLabel":"3","variantLabel":"Large","createdAt":"2025-10-30T22:21:06.000Z","kind":"embedding","contextLength":8192,"inputModalities":["text"],"outputModalities":["embeddings"],"promptPerMillion":0.13,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"text-embedding-3-large is OpenAI's most capable embedding model for both english and non-english tasks. Embeddings are a numerical representation of text that can be used to measure the relatedness between two...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/text-embedding-3-large","url":"https://openrouter.ai/openai/text-embedding-3-large","free":false,"batch":false},{"provider":"OpenRouter","modelId":"openai/text-embedding-3-large:batch","url":"https://openrouter.ai/openai/text-embedding-3-large:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"MTEB English retrieval","score":58,"unit":"%","sourceName":"MTEB","sourceUrl":"https://leaderboard.mteb.org/","measuredAt":"2026-08-27","scope":"Mean retrieval score · leaderboard rank 86"}],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[{"id":"cefdb845-e661-43e5-8120-6ae2ed0fe4e5","title":"Learning Word Embedding","href":"/intel/cefdb845-e661-43e5-8120-6ae2ed0fe4e5","hook":"A clear walkthrough of how free-text words become numeric vectors — from one-hot encoding to learned embeddings — grounding the intuition behind the embeddin…"},{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"}]},{"id":"openai/text-embedding-ada-002","name":"Text Embedding Ada 002","releaseLabel":"Text Embedding Ada 002","generationId":"002","generationLabel":"002","variantLabel":"Text Embedding Ada","createdAt":"2025-10-30T23:09:58.000Z","kind":"embedding","contextLength":8192,"inputModalities":["text"],"outputModalities":["embeddings"],"promptPerMillion":0.09999999999999999,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"text-embedding-ada-002 is OpenAI's legacy text embedding model.","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/text-embedding-ada-002","url":"https://openrouter.ai/openai/text-embedding-ada-002","free":false,"batch":false},{"provider":"OpenRouter","modelId":"openai/text-embedding-ada-002:batch","url":"https://openrouter.ai/openai/text-embedding-ada-002:batch","free":false,"batch":true}],"benchmarks":[],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[{"id":"cefdb845-e661-43e5-8120-6ae2ed0fe4e5","title":"Learning Word Embedding","href":"/intel/cefdb845-e661-43e5-8120-6ae2ed0fe4e5","hook":"A clear walkthrough of how free-text words become numeric vectors — from one-hot encoding to learned embeddings — grounding the intuition behind the embeddin…"},{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"}]}]},{"id":"gpt-transcribe","label":"GPT Transcribe","models":[{"id":"openai/gpt-4o-transcribe","name":"GPT-4o Transcribe","releaseLabel":"GPT-4o Transcribe","generationId":"transcribe","generationLabel":"Transcribe","variantLabel":"4o Transcribe","createdAt":"2026-04-27T23:34:55.000Z","kind":"audio","contextLength":128000,"inputModalities":["audio"],"outputModalities":["transcription"],"promptPerMillion":2.5,"completionPerMillion":10,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"GPT-4o Transcribe is OpenAI's high-quality speech-to-text model built on GPT-4o audio capabilities. It's priced per token (input and output), making it suitable for workflows that benefit from token-level billing transparency.","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"openai/gpt-4o-transcribe","url":"https://openrouter.ai/openai/gpt-4o-transcribe","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]},{"id":"openai/gpt-4o-mini-transcribe","name":"GPT-4o Mini Transcribe","releaseLabel":"GPT-4o Mini Transcribe","generationId":"transcribe","generationLabel":"Transcribe","variantLabel":"4o Mini Transcribe","createdAt":"2026-05-01T17:55:51.000Z","kind":"audio","contextLength":128000,"inputModalities":["audio"],"outputModalities":["transcription"],"promptPerMillion":1.25,"completionPerMillion":5,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"GPT-4o Mini Transcribe is OpenAI's smaller, cost-efficient speech-to-text model built on GPT-4o Mini audio capabilities. It's priced per token (input and output), making it suitable for high-volume transcription workflows that...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"openai/gpt-4o-mini-transcribe","url":"https://openrouter.ai/openai/gpt-4o-mini-transcribe","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]},{"id":"openai/gpt-transcribe-20260805","name":"GPT Transcribe","releaseLabel":"GPT Transcribe","generationId":"transcribe","generationLabel":"Transcribe","variantLabel":"Transcribe","createdAt":"2026-08-05T23:51:37.000Z","kind":"audio","contextLength":0,"inputModalities":["audio"],"outputModalities":["transcription"],"promptPerMillion":4500,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"GPT Transcribe is a high-accuracy speech-to-text model from OpenAI. It is suited for recorded audio, streamed file transcription, and committed Realtime turns, with free-form context, keyword hints, and multiple language...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"openai/gpt-transcribe","url":"https://openrouter.ai/openai/gpt-transcribe","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]}]},{"id":"whisper","label":"Whisper","models":[{"id":"openai/whisper-1","name":"Whisper 1","releaseLabel":"1","generationId":"1","generationLabel":"1","variantLabel":"Base","createdAt":"2026-04-27T23:35:05.000Z","kind":"audio","contextLength":0,"inputModalities":["audio"],"outputModalities":["transcription"],"promptPerMillion":6000,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Whisper is OpenAI's open-source automatic speech recognition model, available via API as `whisper-1`. It supports transcription and translation across 50+ languages from audio files up to 25 MB. Accepts formats...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"openai/whisper-1","url":"https://openrouter.ai/openai/whisper-1","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]},{"id":"openai/whisper-large-v3","name":"Whisper Large V3","releaseLabel":"Large V3","generationId":"v3","generationLabel":"V3","variantLabel":"Large","createdAt":"2026-05-01T13:31:06.000Z","kind":"audio","contextLength":0,"inputModalities":["audio"],"outputModalities":["transcription"],"promptPerMillion":7.5,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Whisper Large V3 is OpenAI's open-source automatic speech recognition model offering both audio transcription and translation. It supports 99+ languages and accepts common audio formats including mp3, mp4, wav, webm,...","huggingFaceId":"openai/whisper-large-v3","routes":[{"provider":"OpenRouter","modelId":"openai/whisper-large-v3","url":"https://openrouter.ai/openai/whisper-large-v3","free":false,"batch":false}],"benchmarks":[{"benchmark":"open-asr-leaderboard · mean wer","score":7.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2023-11-07","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · ami wer","score":15.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2023-11-07","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · earnings22 wer","score":11.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2023-11-07","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · gigaspeech wer","score":10,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2023-11-07","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · librispeech clean wer","score":2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2023-11-07","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · librispeech other wer","score":3.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2023-11-07","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · spgispeech wer","score":2.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2023-11-07","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · tedlium wer","score":3.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2023-11-07","scope":"Registered model evaluation · open-asr-leaderboard"}],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]},{"id":"openai/whisper-large-v3-turbo","name":"Whisper Large V3 Turbo","releaseLabel":"Large V3 Turbo","generationId":"v3","generationLabel":"V3","variantLabel":"Turbo","createdAt":"2026-05-01T13:31:06.000Z","kind":"audio","contextLength":0,"inputModalities":["audio"],"outputModalities":["transcription"],"promptPerMillion":3.33,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Whisper Large V3 Turbo is an optimized version of OpenAI's Whisper Large V3 speech recognition model, designed for speed and cost efficiency. It supports transcription across 99+ languages with a...","huggingFaceId":"openai/whisper-large-v3-turbo","routes":[{"provider":"OpenRouter","modelId":"openai/whisper-large-v3-turbo","url":"https://openrouter.ai/openai/whisper-large-v3-turbo","free":false,"batch":false}],"benchmarks":[{"benchmark":"open-asr-leaderboard · mean wer","score":7.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2024-10-01","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · ami wer","score":16.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2024-10-01","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · earnings22 wer","score":11.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2024-10-01","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · gigaspeech wer","score":10.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2024-10-01","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · librispeech clean wer","score":2.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2024-10-01","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · librispeech other wer","score":4.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2024-10-01","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · spgispeech wer","score":3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2024-10-01","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · tedlium wer","score":3.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2024-10-01","scope":"Registered model evaluation · open-asr-leaderboard"}],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]}]},{"id":"gpt-audio","label":"GPT Audio","models":[{"id":"openai/gpt-audio-mini","name":"GPT Audio Mini","releaseLabel":"Mini","generationId":"audio","generationLabel":"Audio","variantLabel":"Mini","createdAt":"2026-01-19T21:50:19.000Z","kind":"audio","contextLength":128000,"inputModalities":["text","audio"],"outputModalities":["text","audio"],"promptPerMillion":0.6,"completionPerMillion":2.4,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"A cost-efficient version of GPT Audio. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Input is priced at $0.60 per million...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-audio-mini","url":"https://openrouter.ai/openai/gpt-audio-mini","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]},{"id":"openai/gpt-audio","name":"GPT Audio","releaseLabel":"GPT Audio","generationId":"audio","generationLabel":"Audio","variantLabel":"Base","createdAt":"2026-01-19T22:42:49.000Z","kind":"audio","contextLength":128000,"inputModalities":["text","audio"],"outputModalities":["text","audio"],"promptPerMillion":2.5,"completionPerMillion":10,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"The gpt-audio model is OpenAI's first generally available audio model. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Audio is priced...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-audio","url":"https://openrouter.ai/openai/gpt-audio","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]}]},{"id":"gpt-oss","label":"GPT OSS","models":[{"id":"openai/gpt-oss-20b","name":"gpt-oss-20b","releaseLabel":"gpt-oss-20b","generationId":"20b","generationLabel":"20B","variantLabel":"GPT OSS","createdAt":"2025-08-05T17:17:09.000Z","kind":"language","contextLength":131072,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.03,"completionPerMillion":0.13,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","huggingFaceId":"openai/gpt-oss-20b","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-oss-20b","url":"https://openrouter.ai/openai/gpt-oss-20b","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1288,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 238 · 10,669 votes"},{"benchmark":"GPQA","score":56.8,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"Humanity's Last Exam","score":4.2,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"SWE-bench Verified","score":37.4,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"ifstruct-v1.0 · ifstruct v1","score":92,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://www.liquid.ai/blog/ifstruct-v1.0","measuredAt":"2026-06-30","scope":"Registered model evaluation · Liquid AI — IFStruct v1.0 blog (gpt-oss-20b)"},{"benchmark":"gpqa · diamond","score":58.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/6f/c3/6fc3630c-36f0-4ba9-9f34-3c9f73dec76a.json","measuredAt":"2026-04-19","scope":"Registered model evaluation · EvalEval"},{"benchmark":"MMLU-Pro · mmlu pro","score":73.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/ca/ef/caef6dbc-0f07-4491-9009-2b0d26dc9724.json","measuredAt":"2026-06-30","scope":"Registered model evaluation · EvalEval"},{"benchmark":"LEXam · open question","score":32.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://lexam-benchmark.github.io/","measuredAt":"2026-06-02","scope":"Registered model evaluation · LEXam Leaderboard"},{"benchmark":"LEXam · mcq 4 choices","score":40.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://lexam-benchmark.github.io/","measuredAt":"2026-06-02","scope":"Registered model evaluation · LEXam Leaderboard"},{"benchmark":"SWE-bench_Verified · swe bench % resolved","score":53.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://arxiv.org/abs/2508.10925","measuredAt":"2025-08-05","scope":"Registered model evaluation · GPT-OSS Model Card"},{"benchmark":"hle · hle","score":8.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://arxiv.org/abs/2508.10925","measuredAt":"2025-08-05","scope":"Registered model evaluation · GPT-OSS Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]},{"id":"openai/gpt-oss-120b","name":"gpt-oss-120b","releaseLabel":"gpt-oss-120b","generationId":"120b","generationLabel":"120B","variantLabel":"GPT OSS","createdAt":"2025-08-05T17:17:11.000Z","kind":"language","contextLength":131072,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.037,"completionPerMillion":0.16999999999999998,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","huggingFaceId":"openai/gpt-oss-120b","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-oss-120b","url":"https://openrouter.ai/openai/gpt-oss-120b","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":46.4,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"},{"benchmark":"AA Intelligence Index","score":24.1,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-oss-120b","measuredAt":"2026-08-27","scope":"Current public index; highest tested effort configuration"},{"benchmark":"AA Agentic Index","score":13.4,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-oss-120b","measuredAt":"2026-08-27","scope":"Current public agentic composite"},{"benchmark":"SciCode","score":38.9,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-oss-120b","measuredAt":"2026-08-27","scope":"Scientific coding evaluation"},{"benchmark":"AA Omniscience Index","score":-49.3,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-oss-120b","measuredAt":"2026-08-27","scope":"Current public knowledge and hallucination index"},{"benchmark":"MLCR overall","score":1.1,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-oss-120b","measuredAt":"2026-08-27","scope":"Machine-learning code reasoning evaluation"},{"benchmark":"GDPval-AA normalized","score":15.1,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-oss-120b","measuredAt":"2026-08-27","scope":"Normalized real-world knowledge-work evaluation"},{"benchmark":"ITBench SRE","score":5.6,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-oss-120b","measuredAt":"2026-08-27","scope":"Site-reliability engineering agent tasks"},{"benchmark":"τ²-bench","score":65.8,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-oss-120b","measuredAt":"2026-08-27","scope":"Tool-agent benchmark across supported domains"},{"benchmark":"τ²-bench Banking","score":12.8,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-oss-120b","measuredAt":"2026-08-27","scope":"Tool-agent benchmark in the banking domain"},{"benchmark":"Terminal-Bench Hard","score":23.5,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-oss-120b","measuredAt":"2026-08-27","scope":"Hard terminal-agent tasks"},{"benchmark":"Terminal-Bench 2.1","score":26.2,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-oss-120b","measuredAt":"2026-08-27","scope":"Terminal-agent task completion"},{"benchmark":"LCR","score":51,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-oss-120b","measuredAt":"2026-08-27","scope":"Long-context reasoning evaluation"},{"benchmark":"IFBench","score":69,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-oss-120b","measuredAt":"2026-08-27","scope":"Instruction-following evaluation"},{"benchmark":"Humanity's Last Exam","score":19.6,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-oss-120b","measuredAt":"2026-08-27","scope":"Broad expert-level knowledge evaluation"},{"benchmark":"GPQA","score":78.2,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-oss-120b","measuredAt":"2026-08-27","scope":"Graduate-level science questions"},{"benchmark":"CritPt","score":1.1,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-oss-120b","measuredAt":"2026-08-27","scope":"Critical-point reasoning evaluation"},{"benchmark":"APEX Agents","score":3.1,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-oss-120b","measuredAt":"2026-08-27","scope":"Professional agent task evaluation"},{"benchmark":"LiveCodeBench","score":87.8,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-oss-120b","measuredAt":"2026-08-27","scope":"Contamination-resistant coding tasks"},{"benchmark":"AIME 2025","score":93.4,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-oss-120b","measuredAt":"2026-08-27","scope":"Competition mathematics evaluation"},{"benchmark":"Enterprise Ops Gym","score":25.5,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-oss-120b","measuredAt":"2026-08-27","scope":"Enterprise operations agent tasks"},{"benchmark":"LMArena overall","score":1366,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 172 · 30,761 votes"},{"benchmark":"SWE-bench Pro","score":16.2,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"SWE-bench Verified","score":47.9,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"gpqa · diamond","score":80.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/5d/3e/5d3e8832-c770-4796-957b-dfaefbd3bb2a.json","measuredAt":"2026-04-16","scope":"Registered model evaluation · EvalEval"},{"benchmark":"MMLU-Pro · mmlu pro","score":80.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/e4/8d/e48dbb1a-9e1a-4c99-abbf-cb32942b48e7.json","measuredAt":"2026-06-30","scope":"Registered model evaluation · EvalEval"},{"benchmark":"LEXam · open question","score":51.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://lexam-benchmark.github.io/","measuredAt":"2026-06-02","scope":"Registered model evaluation · LEXam Leaderboard"},{"benchmark":"LEXam · mcq 4 choices","score":47.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://lexam-benchmark.github.io/","measuredAt":"2026-06-02","scope":"Registered model evaluation · LEXam Leaderboard"},{"benchmark":"SWE-bench_Verified · swe bench % resolved","score":52.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://arxiv.org/abs/2508.10925","measuredAt":"2025-08-05","scope":"Registered model evaluation · GPT-OSS Model Card"},{"benchmark":"hle · hle","score":11.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://arxiv.org/abs/2508.10925","measuredAt":"2025-08-05","scope":"Registered model evaluation · GPT-OSS Model Card"},{"benchmark":"SWE-bench_Pro · SWE Bench Pro","score":16.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://scale.com/leaderboard/swe_bench_pro_public","measuredAt":"2026-02-28","scope":"Registered model evaluation · SWE-Bench Pro official evaluation results"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]}]},{"id":"gpt-oss-safeguard","label":"GPT OSS Safeguard","models":[{"id":"openai/gpt-oss-safeguard-20b","name":"gpt-oss-safeguard-20b","releaseLabel":"gpt-oss-safeguard-20b","generationId":"initial","generationLabel":"Initial","variantLabel":"20B","createdAt":"2025-10-29T15:47:16.000Z","kind":"language","contextLength":131072,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.075,"completionPerMillion":0.3,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"gpt-oss-safeguard-20b is a safety reasoning model from OpenAI built upon gpt-oss-20b. This open-weight, 21B-parameter Mixture-of-Experts (MoE) model offers lower latency for safety tasks like content classification, LLM filtering, and trust...","huggingFaceId":"openai/gpt-oss-safeguard-20b","routes":[{"provider":"OpenRouter","modelId":"openai/gpt-oss-safeguard-20b","url":"https://openrouter.ai/openai/gpt-oss-safeguard-20b","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]}]},{"id":"sora","label":"Sora","models":[{"id":"openai/sora-2-pro-20260320","name":"Sora 2 Pro","releaseLabel":"2 Pro","generationId":"2","generationLabel":"2","variantLabel":"Pro","createdAt":"2026-03-23T14:52:01.000Z","kind":"video","contextLength":0,"inputModalities":["text","image"],"outputModalities":["video"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"OpenAI's flagship video generation model, delivering production-quality video with physics-accurate motion, synchronized audio, and world-state persistence across shots. Sora 2 Pro follows intricate multi-shot instructions while maintaining consistent spatial relationships...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"openai/sora-2-pro","url":"https://openrouter.ai/openai/sora-2-pro","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Video Arena","url":"https://artificialanalysis.ai/video/leaderboard/text-to-video","note":"Blind preference ranking for video generation"}],"intel":[{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"},{"id":"cc857066-48cb-40e3-983e-09b179adb31b","title":"Codex Orange Book","href":"/intel/cc857066-48cb-40e3-983e-09b179adb31b","hook":"A practical field guide to OpenAI Codex agentic workflows for the GPT-5.5 era — prompts and patterns, not marketing. Useful if you are standing up Codex-styl…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"}]}]}]},{"id":"google","name":"Google","domain":"deepmind.google","major":true,"modelCount":36,"series":[{"id":"gemini","label":"Gemini","models":[{"id":"google/gemini-2.5-pro-preview-03-25","name":"Gemini 2.5 Pro Preview 05-06","releaseLabel":"2.5 Pro Preview 05-06","generationId":"2.5","generationLabel":"2.5","variantLabel":"Pro Preview · 05-06","createdAt":"2025-05-07T00:41:53.000Z","kind":"multimodal","contextLength":1048576,"inputModalities":["text","image","file","audio","video"],"outputModalities":["text"],"promptPerMillion":1.25,"completionPerMillion":10,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"google/gemini-2.5-pro-preview-05-06","url":"https://openrouter.ai/google/gemini-2.5-pro-preview-05-06","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":80.2,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2025_04_25.csv","measuredAt":"2025-04-25","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1457,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 32 · 124,807 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"5b24252c-82d9-48ef-bf7f-fc3593c9bb49","title":"The non-technical PM’s guide to building with Cursor | Zevi Arnovitz (Meta)","href":"/intel/5b24252c-82d9-48ef-bf7f-fc3593c9bb49","hook":"Breaks down a repeatable Cursor workflow — routing tasks across Claude and Gemini, automating prompts with slash commands, and having models peer-review each…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"}]},{"id":"google/gemini-2.5-pro-preview-06-05","name":"Gemini 2.5 Pro Preview 06-05","releaseLabel":"2.5 Pro Preview 06-05","generationId":"2.5","generationLabel":"2.5","variantLabel":"Pro Preview","createdAt":"2025-06-05T15:27:37.000Z","kind":"multimodal","contextLength":1048576,"inputModalities":["file","image","text","audio"],"outputModalities":["text"],"promptPerMillion":1.25,"completionPerMillion":10,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"google/gemini-2.5-pro-preview","url":"https://openrouter.ai/google/gemini-2.5-pro-preview","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":80.2,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2025_04_25.csv","measuredAt":"2025-04-25","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1457,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 32 · 124,807 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"5b24252c-82d9-48ef-bf7f-fc3593c9bb49","title":"The non-technical PM’s guide to building with Cursor | Zevi Arnovitz (Meta)","href":"/intel/5b24252c-82d9-48ef-bf7f-fc3593c9bb49","hook":"Breaks down a repeatable Cursor workflow — routing tasks across Claude and Gemini, automating prompts with slash commands, and having models peer-review each…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"}]},{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","releaseLabel":"2.5 Pro","generationId":"2.5","generationLabel":"2.5","variantLabel":"Pro","createdAt":"2025-06-17T14:12:24.000Z","kind":"multimodal","contextLength":1048576,"inputModalities":["text","image","file","audio","video"],"outputModalities":["text"],"promptPerMillion":1.25,"completionPerMillion":10,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"google/gemini-2.5-pro","url":"https://openrouter.ai/google/gemini-2.5-pro","free":false,"batch":false},{"provider":"OpenRouter","modelId":"google/gemini-2.5-pro:batch","url":"https://openrouter.ai/google/gemini-2.5-pro:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"LiveBench overall","score":80.2,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2025_04_25.csv","measuredAt":"2025-04-25","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1457,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 32 · 124,807 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"5b24252c-82d9-48ef-bf7f-fc3593c9bb49","title":"The non-technical PM’s guide to building with Cursor | Zevi Arnovitz (Meta)","href":"/intel/5b24252c-82d9-48ef-bf7f-fc3593c9bb49","hook":"Breaks down a repeatable Cursor workflow — routing tasks across Claude and Gemini, automating prompts with slash commands, and having models peer-review each…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"}]},{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","releaseLabel":"2.5 Flash","generationId":"2.5","generationLabel":"2.5","variantLabel":"Flash","createdAt":"2025-06-17T15:01:28.000Z","kind":"multimodal","contextLength":1048576,"inputModalities":["file","image","text","audio","video"],"outputModalities":["text"],"promptPerMillion":0.3,"completionPerMillion":2.5,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"google/gemini-2.5-flash","url":"https://openrouter.ai/google/gemini-2.5-flash","free":false,"batch":false},{"provider":"OpenRouter","modelId":"google/gemini-2.5-flash:batch","url":"https://openrouter.ai/google/gemini-2.5-flash:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"LiveBench overall","score":52.3,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1417,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 111 · 124,731 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"5b24252c-82d9-48ef-bf7f-fc3593c9bb49","title":"The non-technical PM’s guide to building with Cursor | Zevi Arnovitz (Meta)","href":"/intel/5b24252c-82d9-48ef-bf7f-fc3593c9bb49","hook":"Breaks down a repeatable Cursor workflow — routing tasks across Claude and Gemini, automating prompts with slash commands, and having models peer-review each…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"}]},{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash Lite","releaseLabel":"2.5 Flash Lite","generationId":"2.5","generationLabel":"2.5","variantLabel":"Flash Lite","createdAt":"2025-07-22T16:04:36.000Z","kind":"multimodal","contextLength":1048576,"inputModalities":["text","image","file","audio","video"],"outputModalities":["text"],"promptPerMillion":0.09999999999999999,"completionPerMillion":0.39999999999999997,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"google/gemini-2.5-flash-lite","url":"https://openrouter.ai/google/gemini-2.5-flash-lite","free":false,"batch":false},{"provider":"OpenRouter","modelId":"google/gemini-2.5-flash-lite:batch","url":"https://openrouter.ai/google/gemini-2.5-flash-lite:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"LiveBench overall","score":41.5,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1379,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 153 · 47,448 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"5b24252c-82d9-48ef-bf7f-fc3593c9bb49","title":"The non-technical PM’s guide to building with Cursor | Zevi Arnovitz (Meta)","href":"/intel/5b24252c-82d9-48ef-bf7f-fc3593c9bb49","hook":"Breaks down a repeatable Cursor workflow — routing tasks across Claude and Gemini, automating prompts with slash commands, and having models peer-review each…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"}]},{"id":"google/gemini-3-flash-preview-20251217","name":"Gemini 3 Flash Preview","releaseLabel":"3 Flash Preview","generationId":"3","generationLabel":"3","variantLabel":"Flash Preview","createdAt":"2025-12-17T15:57:58.000Z","kind":"multimodal","contextLength":1048576,"inputModalities":["text","image","file","audio","video"],"outputModalities":["text"],"promptPerMillion":0.5,"completionPerMillion":3,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"google/gemini-3-flash-preview","url":"https://openrouter.ai/google/gemini-3-flash-preview","free":false,"batch":false},{"provider":"OpenRouter","modelId":"google/gemini-3-flash-preview:batch","url":"https://openrouter.ai/google/gemini-3-flash-preview:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"LiveBench overall","score":54.4,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1466,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 25 · 30,852 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"5b24252c-82d9-48ef-bf7f-fc3593c9bb49","title":"The non-technical PM’s guide to building with Cursor | Zevi Arnovitz (Meta)","href":"/intel/5b24252c-82d9-48ef-bf7f-fc3593c9bb49","hook":"Breaks down a repeatable Cursor workflow — routing tasks across Claude and Gemini, automating prompts with slash commands, and having models peer-review each…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"}]},{"id":"google/gemini-3.1-pro-preview-20260219","name":"Gemini 3.1 Pro Preview","releaseLabel":"3.1 Pro Preview","generationId":"3.1","generationLabel":"3.1","variantLabel":"Pro Preview","createdAt":"2026-02-19T14:00:27.000Z","kind":"multimodal","contextLength":1048576,"inputModalities":["audio","file","image","text","video"],"outputModalities":["text"],"promptPerMillion":2,"completionPerMillion":12,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"google/gemini-3.1-pro-preview","url":"https://openrouter.ai/google/gemini-3.1-pro-preview","free":false,"batch":false},{"provider":"OpenRouter","modelId":"google/gemini-3.1-pro-preview:batch","url":"https://openrouter.ai/google/gemini-3.1-pro-preview:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"LiveBench overall","score":78,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1480,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 12 · 99,182 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"5b24252c-82d9-48ef-bf7f-fc3593c9bb49","title":"The non-technical PM’s guide to building with Cursor | Zevi Arnovitz (Meta)","href":"/intel/5b24252c-82d9-48ef-bf7f-fc3593c9bb49","hook":"Breaks down a repeatable Cursor workflow — routing tasks across Claude and Gemini, automating prompts with slash commands, and having models peer-review each…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"}]},{"id":"google/gemini-3.1-pro-preview-customtools-20260219","name":"Gemini 3.1 Pro Preview Custom Tools","releaseLabel":"3.1 Pro Preview Custom Tools","generationId":"3.1","generationLabel":"3.1","variantLabel":"Pro Preview Customtools","createdAt":"2026-02-25T18:58:43.000Z","kind":"multimodal","contextLength":1048576,"inputModalities":["text","audio","image","video","file"],"outputModalities":["text"],"promptPerMillion":2,"completionPerMillion":12,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Gemini 3.1 Pro Preview Custom Tools is a variant of Gemini 3.1 Pro that improves tool selection behavior by preventing overuse of a general bash tool when more efficient third-party...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"google/gemini-3.1-pro-preview-customtools","url":"https://openrouter.ai/google/gemini-3.1-pro-preview-customtools","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":78,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1480,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 12 · 99,182 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"5b24252c-82d9-48ef-bf7f-fc3593c9bb49","title":"The non-technical PM’s guide to building with Cursor | Zevi Arnovitz (Meta)","href":"/intel/5b24252c-82d9-48ef-bf7f-fc3593c9bb49","hook":"Breaks down a repeatable Cursor workflow — routing tasks across Claude and Gemini, automating prompts with slash commands, and having models peer-review each…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"}]},{"id":"google/gemini-3.1-flash-lite-preview-20260303","name":"Gemini 3.1 Flash Lite Preview","releaseLabel":"3.1 Flash Lite Preview","generationId":"3.1","generationLabel":"3.1","variantLabel":"Flash Lite Preview","createdAt":"2026-03-03T04:37:53.000Z","kind":"multimodal","contextLength":1048576,"inputModalities":["text","image","video","file","audio"],"outputModalities":["text"],"promptPerMillion":0.25,"completionPerMillion":1.5,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Gemini 3.1 Flash Lite Preview is Google's high-efficiency model optimized for high-volume use cases. It outperforms Gemini 2.5 Flash Lite on overall quality and approaches Gemini 2.5 Flash performance across...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"google/gemini-3.1-flash-lite-preview","url":"https://openrouter.ai/google/gemini-3.1-flash-lite-preview","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":62.1,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1415,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 116 · 60,594 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"5b24252c-82d9-48ef-bf7f-fc3593c9bb49","title":"The non-technical PM’s guide to building with Cursor | Zevi Arnovitz (Meta)","href":"/intel/5b24252c-82d9-48ef-bf7f-fc3593c9bb49","hook":"Breaks down a repeatable Cursor workflow — routing tasks across Claude and Gemini, automating prompts with slash commands, and having models peer-review each…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"}]},{"id":"google/gemini-3.1-flash-lite-20260507","name":"Gemini 3.1 Flash Lite","releaseLabel":"3.1 Flash Lite","generationId":"3.1","generationLabel":"3.1","variantLabel":"Flash Lite","createdAt":"2026-05-07T15:47:08.000Z","kind":"multimodal","contextLength":1048576,"inputModalities":["text","image","video","file","audio"],"outputModalities":["text"],"promptPerMillion":0.25,"completionPerMillion":1.5,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"google/gemini-3.1-flash-lite","url":"https://openrouter.ai/google/gemini-3.1-flash-lite","free":false,"batch":false},{"provider":"OpenRouter","modelId":"google/gemini-3.1-flash-lite:batch","url":"https://openrouter.ai/google/gemini-3.1-flash-lite:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"LiveBench overall","score":62.1,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1415,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 116 · 60,594 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"5b24252c-82d9-48ef-bf7f-fc3593c9bb49","title":"The non-technical PM’s guide to building with Cursor | Zevi Arnovitz (Meta)","href":"/intel/5b24252c-82d9-48ef-bf7f-fc3593c9bb49","hook":"Breaks down a repeatable Cursor workflow — routing tasks across Claude and Gemini, automating prompts with slash commands, and having models peer-review each…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"}]},{"id":"google/gemini-3.5-flash-20260519","name":"Gemini 3.5 Flash","releaseLabel":"3.5 Flash","generationId":"3.5","generationLabel":"3.5","variantLabel":"Flash","createdAt":"2026-05-19T12:30:00.000Z","kind":"multimodal","contextLength":1048576,"inputModalities":["text","image","video","file","audio"],"outputModalities":["text"],"promptPerMillion":1.5,"completionPerMillion":9,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"google/gemini-3.5-flash","url":"https://openrouter.ai/google/gemini-3.5-flash","free":false,"batch":false},{"provider":"OpenRouter","modelId":"google/gemini-3.5-flash:batch","url":"https://openrouter.ai/google/gemini-3.5-flash:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"LiveBench overall","score":75.4,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1483,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 10 · 30,004 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"5b24252c-82d9-48ef-bf7f-fc3593c9bb49","title":"The non-technical PM’s guide to building with Cursor | Zevi Arnovitz (Meta)","href":"/intel/5b24252c-82d9-48ef-bf7f-fc3593c9bb49","hook":"Breaks down a repeatable Cursor workflow — routing tasks across Claude and Gemini, automating prompts with slash commands, and having models peer-review each…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"}]},{"id":"google/gemini-3.5-flash-lite-20260721","name":"Gemini 3.5 Flash Lite","releaseLabel":"3.5 Flash Lite","generationId":"3.5","generationLabel":"3.5","variantLabel":"Flash Lite","createdAt":"2026-07-21T15:12:06.000Z","kind":"multimodal","contextLength":1048576,"inputModalities":["text","image","video","file","audio"],"outputModalities":["text"],"promptPerMillion":0.3,"completionPerMillion":2.5,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"google/gemini-3.5-flash-lite","url":"https://openrouter.ai/google/gemini-3.5-flash-lite","free":false,"batch":false},{"provider":"OpenRouter","modelId":"google/gemini-3.5-flash-lite:batch","url":"https://openrouter.ai/google/gemini-3.5-flash-lite:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"LiveBench overall","score":63.8,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"AA Intelligence Index","score":37.4,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gemini-3-5-flash-lite","measuredAt":"2026-08-27","scope":"Current public index; highest tested effort configuration"},{"benchmark":"LMArena overall","score":1436,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 73 · 17,985 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"5b24252c-82d9-48ef-bf7f-fc3593c9bb49","title":"The non-technical PM’s guide to building with Cursor | Zevi Arnovitz (Meta)","href":"/intel/5b24252c-82d9-48ef-bf7f-fc3593c9bb49","hook":"Breaks down a repeatable Cursor workflow — routing tasks across Claude and Gemini, automating prompts with slash commands, and having models peer-review each…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"}]},{"id":"google/gemini-3.6-flash-20260721","name":"Gemini 3.6 Flash","releaseLabel":"3.6 Flash","generationId":"3.6","generationLabel":"3.6","variantLabel":"Flash","createdAt":"2026-07-21T15:12:13.000Z","kind":"multimodal","contextLength":1048576,"inputModalities":["text","image","video","file","audio"],"outputModalities":["text"],"promptPerMillion":0.75,"completionPerMillion":3.75,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"google/gemini-3.6-flash","url":"https://openrouter.ai/google/gemini-3.6-flash","free":false,"batch":false},{"provider":"OpenRouter","modelId":"google/gemini-3.6-flash:batch","url":"https://openrouter.ai/google/gemini-3.6-flash:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"LiveBench overall","score":74.5,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1476,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 16 · 18,018 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"5b24252c-82d9-48ef-bf7f-fc3593c9bb49","title":"The non-technical PM’s guide to building with Cursor | Zevi Arnovitz (Meta)","href":"/intel/5b24252c-82d9-48ef-bf7f-fc3593c9bb49","hook":"Breaks down a repeatable Cursor workflow — routing tasks across Claude and Gemini, automating prompts with slash commands, and having models peer-review each…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"}]},{"id":"google/gemini-3.7-flash-20260813","name":"Gemini 3.7 Flash","releaseLabel":"3.7 Flash","generationId":"3.7","generationLabel":"3.7","variantLabel":"Flash","createdAt":"2026-08-13T17:03:01.000Z","kind":"multimodal","contextLength":1048576,"inputModalities":["text","image","video","file","audio"],"outputModalities":["text"],"promptPerMillion":0.375,"completionPerMillion":1.875,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Gemini 3.7 Flash is a multimodal model from Google for fast agentic workflows, coding, and complex multi-step reasoning. It is designed for tasks that require responsive performance and reliable multi-step...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"google/gemini-3.7-flash","url":"https://openrouter.ai/google/gemini-3.7-flash","free":false,"batch":false},{"provider":"OpenRouter","modelId":"google/gemini-3.7-flash:batch","url":"https://openrouter.ai/google/gemini-3.7-flash:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"LiveBench overall","score":79.9,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"AA Intelligence Index","score":56,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gemini-3-7-flash","measuredAt":"2026-08-27","scope":"Current public index; highest tested effort configuration"},{"benchmark":"LMArena overall","score":1490,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 6 · 5,718 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"5b24252c-82d9-48ef-bf7f-fc3593c9bb49","title":"The non-technical PM’s guide to building with Cursor | Zevi Arnovitz (Meta)","href":"/intel/5b24252c-82d9-48ef-bf7f-fc3593c9bb49","hook":"Breaks down a repeatable Cursor workflow — routing tasks across Claude and Gemini, automating prompts with slash commands, and having models peer-review each…"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"}]}]},{"id":"gemini-image","label":"Gemini Image","models":[{"id":"google/gemini-2.5-flash-image","name":"Nano Banana (Gemini 2.5 Flash Image)","releaseLabel":"Nano Banana (Gemini 2.5 Flash Image)","generationId":"2.5","generationLabel":"2.5","variantLabel":"Flash","createdAt":"2025-10-07T20:53:51.000Z","kind":"image","contextLength":32768,"inputModalities":["image","text"],"outputModalities":["image","text"],"promptPerMillion":0.3,"completionPerMillion":2.5,"imageOutput":30,"supportsTools":false,"supportsReasoning":false,"description":"Gemini 2.5 Flash Image, a.k.a. \"Nano Banana,\" is now generally available. It is a state of the art image generation model with contextual understanding. It is capable of image generation,...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"google/gemini-2.5-flash-image","url":"https://openrouter.ai/google/gemini-2.5-flash-image","free":false,"batch":false}],"benchmarks":[{"benchmark":"AA Image Arena Elo","score":1188,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/image/leaderboard/text-to-image","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"},{"id":"d6aac06e-b988-4207-8c97-94c17fbe4707","title":"The 100-person AI lab that became Anthropic and Google's secret weapon | Edwin Chen (Surge AI)","href":"/intel/d6aac06e-b988-4207-8c97-94c17fbe4707","hook":"Surge AI supplies the human data, evals, and RL environments behind leading models, and this conversation explains why benchmark-chasing misleads training an…"}]},{"id":"google/gemini-3-pro-image-preview-20251120","name":"Nano Banana Pro (Gemini 3 Pro Image Preview)","releaseLabel":"Nano Banana Pro (Gemini 3 Pro Image Preview)","generationId":"3","generationLabel":"3","variantLabel":"Pro Preview","createdAt":"2025-11-20T15:49:57.000Z","kind":"image","contextLength":65536,"inputModalities":["image","text"],"outputModalities":["image","text"],"promptPerMillion":2,"completionPerMillion":12,"imageOutput":120,"supportsTools":false,"supportsReasoning":true,"description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"google/gemini-3-pro-image-preview","url":"https://openrouter.ai/google/gemini-3-pro-image-preview","free":false,"batch":false}],"benchmarks":[{"benchmark":"AA Image Arena Elo","score":1297,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/image/leaderboard/text-to-image","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"},{"id":"d6aac06e-b988-4207-8c97-94c17fbe4707","title":"The 100-person AI lab that became Anthropic and Google's secret weapon | Edwin Chen (Surge AI)","href":"/intel/d6aac06e-b988-4207-8c97-94c17fbe4707","hook":"Surge AI supplies the human data, evals, and RL environments behind leading models, and this conversation explains why benchmark-chasing misleads training an…"}]},{"id":"google/gemini-3.1-flash-image-preview-20260226","name":"Nano Banana 2 (Gemini 3.1 Flash Image Preview)","releaseLabel":"Nano Banana 2 (Gemini 3.1 Flash Image Preview)","generationId":"3.1","generationLabel":"3.1","variantLabel":"Flash Preview","createdAt":"2026-02-26T15:25:58.000Z","kind":"image","contextLength":65536,"inputModalities":["image","text"],"outputModalities":["image","text"],"promptPerMillion":0.5,"completionPerMillion":3,"imageOutput":60,"supportsTools":false,"supportsReasoning":true,"description":"Gemini 3.1 Flash Image Preview, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"google/gemini-3.1-flash-image-preview","url":"https://openrouter.ai/google/gemini-3.1-flash-image-preview","free":false,"batch":false}],"benchmarks":[{"benchmark":"AA Image Arena Elo","score":1322,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/image/leaderboard/text-to-image","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"},{"id":"d6aac06e-b988-4207-8c97-94c17fbe4707","title":"The 100-person AI lab that became Anthropic and Google's secret weapon | Edwin Chen (Surge AI)","href":"/intel/d6aac06e-b988-4207-8c97-94c17fbe4707","hook":"Surge AI supplies the human data, evals, and RL environments behind leading models, and this conversation explains why benchmark-chasing misleads training an…"}]},{"id":"google/gemini-3-pro-image-20260528","name":"Nano Banana Pro (Gemini 3 Pro Image)","releaseLabel":"Nano Banana Pro (Gemini 3 Pro Image)","generationId":"3","generationLabel":"3","variantLabel":"Pro","createdAt":"2026-06-18T03:40:54.000Z","kind":"image","contextLength":131072,"inputModalities":["image","text"],"outputModalities":["image","text"],"promptPerMillion":2,"completionPerMillion":12,"imageOutput":120,"supportsTools":true,"supportsReasoning":true,"description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"google/gemini-3-pro-image","url":"https://openrouter.ai/google/gemini-3-pro-image","free":false,"batch":false}],"benchmarks":[{"benchmark":"AA Image Arena Elo","score":1297,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/image/leaderboard/text-to-image","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"},{"id":"d6aac06e-b988-4207-8c97-94c17fbe4707","title":"The 100-person AI lab that became Anthropic and Google's secret weapon | Edwin Chen (Surge AI)","href":"/intel/d6aac06e-b988-4207-8c97-94c17fbe4707","hook":"Surge AI supplies the human data, evals, and RL environments behind leading models, and this conversation explains why benchmark-chasing misleads training an…"}]},{"id":"google/gemini-3.1-flash-image-20260528","name":"Nano Banana 2 (Gemini 3.1 Flash Image)","releaseLabel":"Nano Banana 2 (Gemini 3.1 Flash Image)","generationId":"3.1","generationLabel":"3.1","variantLabel":"Flash","createdAt":"2026-06-18T03:41:05.000Z","kind":"image","contextLength":131072,"inputModalities":["image","text"],"outputModalities":["image","text"],"promptPerMillion":0.5,"completionPerMillion":3,"imageOutput":60,"supportsTools":false,"supportsReasoning":true,"description":"Gemini 3.1 Flash Image, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines advanced...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"google/gemini-3.1-flash-image","url":"https://openrouter.ai/google/gemini-3.1-flash-image","free":false,"batch":false}],"benchmarks":[{"benchmark":"AA Image Arena Elo","score":1322,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/image/leaderboard/text-to-image","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"},{"id":"d6aac06e-b988-4207-8c97-94c17fbe4707","title":"The 100-person AI lab that became Anthropic and Google's secret weapon | Edwin Chen (Surge AI)","href":"/intel/d6aac06e-b988-4207-8c97-94c17fbe4707","hook":"Surge AI supplies the human data, evals, and RL environments behind leading models, and this conversation explains why benchmark-chasing misleads training an…"}]},{"id":"google/gemini-3.1-flash-lite-image-20260630","name":"Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image)","releaseLabel":"Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image)","generationId":"3.1","generationLabel":"3.1","variantLabel":"Flash Lite","createdAt":"2026-06-30T16:33:45.000Z","kind":"image","contextLength":65536,"inputModalities":["image","text"],"outputModalities":["image","text"],"promptPerMillion":0.25,"completionPerMillion":1.5,"imageOutput":30,"supportsTools":false,"supportsReasoning":true,"description":"Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image) is Google's fastest, most cost-efficient Gemini image model, built for high-velocity developer pipelines and rapid-fire visual exploration. It delivers text-to-image generation...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"google/gemini-3.1-flash-lite-image","url":"https://openrouter.ai/google/gemini-3.1-flash-lite-image","free":false,"batch":false}],"benchmarks":[{"benchmark":"AA Image Arena Elo","score":1289,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/image/leaderboard/text-to-image","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"},{"id":"d6aac06e-b988-4207-8c97-94c17fbe4707","title":"The 100-person AI lab that became Anthropic and Google's secret weapon | Edwin Chen (Surge AI)","href":"/intel/d6aac06e-b988-4207-8c97-94c17fbe4707","hook":"Surge AI supplies the human data, evals, and RL environments behind leading models, and this conversation explains why benchmark-chasing misleads training an…"}]}]},{"id":"gemma","label":"Gemma","models":[{"id":"google/gemma-2-27b-it","name":"Gemma 2 27B","releaseLabel":"2 27B","generationId":"2","generationLabel":"2","variantLabel":"27B","createdAt":"2024-07-13T00:00:00.000Z","kind":"language","contextLength":8192,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.65,"completionPerMillion":0.65,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Gemma 2 27B by Google is an open model built from the same research and technology used to create the [Gemini models](/models?q=gemini). Gemma models are well-suited for a variety of...","huggingFaceId":"google/gemma-2-27b-it","routes":[{"provider":"OpenRouter","modelId":"google/gemma-2-27b-it","url":"https://openrouter.ai/google/gemma-2-27b-it","free":false,"batch":false}],"benchmarks":[{"benchmark":"gpqa · main","score":30.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/f9/ff/f9ff11dc-90ee-4cca-abb9-72c6786c30d1.json","measuredAt":"2026-06-30","scope":"Registered model evaluation · EvalEval"},{"benchmark":"gpqa · diamond","score":40.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/d3/1e/d31ebd82-94b8-4d63-be48-0e511171e541.json","measuredAt":"2026-04-16","scope":"Registered model evaluation · EvalEval"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"},{"id":"d6aac06e-b988-4207-8c97-94c17fbe4707","title":"The 100-person AI lab that became Anthropic and Google's secret weapon | Edwin Chen (Surge AI)","href":"/intel/d6aac06e-b988-4207-8c97-94c17fbe4707","hook":"Surge AI supplies the human data, evals, and RL environments behind leading models, and this conversation explains why benchmark-chasing misleads training an…"}]},{"id":"google/gemma-3-27b-it","name":"Gemma 3 27B","releaseLabel":"3 27B","generationId":"3","generationLabel":"3","variantLabel":"27B","createdAt":"2025-03-12T05:12:39.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0.08,"completionPerMillion":0.44999999999999996,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","huggingFaceId":"google/gemma-3-27b-it","routes":[{"provider":"OpenRouter","modelId":"google/gemma-3-27b-it","url":"https://openrouter.ai/google/gemma-3-27b-it","free":false,"batch":false}],"benchmarks":[{"benchmark":"SWE-bench Pro","score":11.4,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"gpqa · main","score":40.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/eb/66/eb66d8ad-546c-436a-9d76-163327b02ee2.json","measuredAt":"2026-06-30","scope":"Registered model evaluation · EvalEval"},{"benchmark":"gpqa · diamond","score":41.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/bf/d8/bfd80231-262a-4ec1-b46b-294739ae65cc.json","measuredAt":"2026-04-18","scope":"Registered model evaluation · EvalEval"},{"benchmark":"SWE-bench_Pro · SWE Bench Pro","score":11.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://scale.com/leaderboard/swe_bench_pro_public","measuredAt":"2026-02-28","scope":"Registered model evaluation · SWE-Bench Pro official evaluation results"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"},{"id":"d6aac06e-b988-4207-8c97-94c17fbe4707","title":"The 100-person AI lab that became Anthropic and Google's secret weapon | Edwin Chen (Surge AI)","href":"/intel/d6aac06e-b988-4207-8c97-94c17fbe4707","hook":"Surge AI supplies the human data, evals, and RL environments behind leading models, and this conversation explains why benchmark-chasing misleads training an…"}]},{"id":"google/gemma-3-12b-it","name":"Gemma 3 12B","releaseLabel":"3 12B","generationId":"3","generationLabel":"3","variantLabel":"12B","createdAt":"2025-03-13T21:50:25.000Z","kind":"multimodal","contextLength":131072,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0.049999999999999996,"completionPerMillion":0.15,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","huggingFaceId":"google/gemma-3-12b-it","routes":[{"provider":"OpenRouter","modelId":"google/gemma-3-12b-it","url":"https://openrouter.ai/google/gemma-3-12b-it","free":false,"batch":false}],"benchmarks":[{"benchmark":"gpqa · main","score":36.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/e4/d2/e4d2c8c5-bcc6-4409-bda8-c955f7e01199.json","measuredAt":"2026-06-30","scope":"Registered model evaluation · EvalEval"},{"benchmark":"gpqa · diamond","score":33.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/34/c3/34c3c055-42ae-4e43-afbe-1b9eb7234c76.json","measuredAt":"2026-04-17","scope":"Registered model evaluation · EvalEval"},{"benchmark":"LEXam · open question","score":41.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://lexam-benchmark.github.io/","measuredAt":"2026-06-02","scope":"Registered model evaluation · LEXam Leaderboard"},{"benchmark":"LEXam · mcq 4 choices","score":29.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://lexam-benchmark.github.io/","measuredAt":"2026-06-02","scope":"Registered model evaluation · LEXam Leaderboard"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"},{"id":"d6aac06e-b988-4207-8c97-94c17fbe4707","title":"The 100-person AI lab that became Anthropic and Google's secret weapon | Edwin Chen (Surge AI)","href":"/intel/d6aac06e-b988-4207-8c97-94c17fbe4707","hook":"Surge AI supplies the human data, evals, and RL environments behind leading models, and this conversation explains why benchmark-chasing misleads training an…"}]},{"id":"google/gemma-3-4b-it","name":"Gemma 3 4B","releaseLabel":"3 4B","generationId":"3","generationLabel":"3","variantLabel":"4B","createdAt":"2025-03-13T22:38:30.000Z","kind":"multimodal","contextLength":131072,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0.049999999999999996,"completionPerMillion":0.09999999999999999,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","huggingFaceId":"google/gemma-3-4b-it","routes":[{"provider":"OpenRouter","modelId":"google/gemma-3-4b-it","url":"https://openrouter.ai/google/gemma-3-4b-it","free":false,"batch":false}],"benchmarks":[{"benchmark":"gpqa · main","score":16.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/0a/fe/0afe633a-9470-4a27-9d66-21a55710d768.json","measuredAt":"2026-06-30","scope":"Registered model evaluation · EvalEval"},{"benchmark":"gpqa · diamond","score":36.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/71/37/7137a16e-89a7-4188-bb8c-7571b42cfe53.json","measuredAt":"2026-04-18","scope":"Registered model evaluation · EvalEval"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"},{"id":"d6aac06e-b988-4207-8c97-94c17fbe4707","title":"The 100-person AI lab that became Anthropic and Google's secret weapon | Edwin Chen (Surge AI)","href":"/intel/d6aac06e-b988-4207-8c97-94c17fbe4707","hook":"Surge AI supplies the human data, evals, and RL environments behind leading models, and this conversation explains why benchmark-chasing misleads training an…"}]},{"id":"google/gemma-4-31b-it-20260402","name":"Gemma 4 31B","releaseLabel":"4 31B","generationId":"4","generationLabel":"4","variantLabel":"31B","createdAt":"2026-04-02T16:48:06.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["image","text","video"],"outputModalities":["text"],"promptPerMillion":0.09,"completionPerMillion":0.33999999999999997,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","huggingFaceId":"google/gemma-4-31B-it","routes":[{"provider":"OpenRouter","modelId":"google/gemma-4-31b-it","url":"https://openrouter.ai/google/gemma-4-31b-it","free":false,"batch":false},{"provider":"OpenRouter","modelId":"google/gemma-4-31b-it:free","url":"https://openrouter.ai/google/gemma-4-31b-it:free","free":true,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1442,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 58 · 5,899 votes"},{"benchmark":"MMMU_Pro · mmmu pro vision","score":76.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/google/gemma-4-31B-it","measuredAt":"2026-05-12","scope":"Registered model evaluation · Model Card"},{"benchmark":"ifstruct-v1.0 · ifstruct v1","score":95.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://www.liquid.ai/blog/ifstruct-v1.0","measuredAt":"2026-06-30","scope":"Registered model evaluation · Liquid AI — IFStruct v1.0 blog (gemma-4-31B-it)"},{"benchmark":"ParseBench · mean","score":62.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/llamaindex/ParseBench","measuredAt":"2026-04-17","scope":"Registered model evaluation · ParseBench"},{"benchmark":"ParseBench · text content","score":89.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/llamaindex/ParseBench","measuredAt":"2026-04-17","scope":"Registered model evaluation · ParseBench"},{"benchmark":"ParseBench · text formatting","score":69.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/llamaindex/ParseBench","measuredAt":"2026-04-17","scope":"Registered model evaluation · ParseBench"},{"benchmark":"ParseBench · layout","score":57.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/llamaindex/ParseBench","measuredAt":"2026-04-17","scope":"Registered model evaluation · ParseBench"},{"benchmark":"ParseBench · chart","score":15,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/llamaindex/ParseBench","measuredAt":"2026-04-17","scope":"Registered model evaluation · ParseBench"},{"benchmark":"ParseBench · table","score":80.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/llamaindex/ParseBench","measuredAt":"2026-04-17","scope":"Registered model evaluation · ParseBench"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"},{"id":"d6aac06e-b988-4207-8c97-94c17fbe4707","title":"The 100-person AI lab that became Anthropic and Google's secret weapon | Edwin Chen (Surge AI)","href":"/intel/d6aac06e-b988-4207-8c97-94c17fbe4707","hook":"Surge AI supplies the human data, evals, and RL environments behind leading models, and this conversation explains why benchmark-chasing misleads training an…"}]},{"id":"google/gemma-4-26b-a4b-it-20260403","name":"Gemma 4 26B A4B","releaseLabel":"4 26B A4B","generationId":"4","generationLabel":"4","variantLabel":"26B A4B","createdAt":"2026-04-03T14:53:09.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["image","text","video"],"outputModalities":["text"],"promptPerMillion":0.07,"completionPerMillion":0.33999999999999997,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","huggingFaceId":"google/gemma-4-26B-A4B-it","routes":[{"provider":"OpenRouter","modelId":"google/gemma-4-26b-a4b-it","url":"https://openrouter.ai/google/gemma-4-26b-a4b-it","free":false,"batch":false},{"provider":"OpenRouter","modelId":"google/gemma-4-26b-a4b-it:free","url":"https://openrouter.ai/google/gemma-4-26b-a4b-it:free","free":true,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1435,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 77 · 5,809 votes"},{"benchmark":"MMMU_Pro · mmmu pro vision","score":73.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/google/gemma-4-26B-A4B-it","measuredAt":"2026-05-12","scope":"Registered model evaluation · Model Card"},{"benchmark":"ExtractBench · mean","score":70.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/llamaindex/ExtractBench","measuredAt":"2026-08-26","scope":"Registered model evaluation · ExtractBench"},{"benchmark":"ExtractBench · short","score":77.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/llamaindex/ExtractBench","measuredAt":"2026-08-26","scope":"Registered model evaluation · ExtractBench"},{"benchmark":"ExtractBench · medium","score":59.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/llamaindex/ExtractBench","measuredAt":"2026-08-26","scope":"Registered model evaluation · ExtractBench"},{"benchmark":"ExtractBench · long","score":28.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/llamaindex/ExtractBench","measuredAt":"2026-08-26","scope":"Registered model evaluation · ExtractBench"},{"benchmark":"ParseBench · mean","score":58.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/llamaindex/ParseBench","measuredAt":"2026-04-14","scope":"Registered model evaluation · ParseBench"},{"benchmark":"ParseBench · text content","score":83.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/llamaindex/ParseBench","measuredAt":"2026-04-14","scope":"Registered model evaluation · ParseBench"},{"benchmark":"ParseBench · text formatting","score":65.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/llamaindex/ParseBench","measuredAt":"2026-04-14","scope":"Registered model evaluation · ParseBench"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"},{"id":"d6aac06e-b988-4207-8c97-94c17fbe4707","title":"The 100-person AI lab that became Anthropic and Google's secret weapon | Edwin Chen (Surge AI)","href":"/intel/d6aac06e-b988-4207-8c97-94c17fbe4707","hook":"Surge AI supplies the human data, evals, and RL environments behind leading models, and this conversation explains why benchmark-chasing misleads training an…"}]}]},{"id":"gemini-embedding","label":"Gemini Embedding","models":[{"id":"google/gemini-embedding-001","name":"Gemini Embedding 001","releaseLabel":"001","generationId":"001","generationLabel":"001","variantLabel":"Base","createdAt":"2025-10-31T20:43:30.000Z","kind":"embedding","contextLength":20000,"inputModalities":["text"],"outputModalities":["embeddings"],"promptPerMillion":0.15,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"gemini-embedding-001 provides a unified cutting edge experience across domains, including science, legal, finance, and coding. This embedding model has consistently held a top spot on the Massive Text Embedding Benchmark...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"google/gemini-embedding-001","url":"https://openrouter.ai/google/gemini-embedding-001","free":false,"batch":false}],"benchmarks":[{"benchmark":"MTEB English retrieval","score":64.4,"unit":"%","sourceName":"MTEB","sourceUrl":"https://leaderboard.mteb.org/","measuredAt":"2026-08-27","scope":"Mean retrieval score · leaderboard rank 9"}],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"},{"id":"d6aac06e-b988-4207-8c97-94c17fbe4707","title":"The 100-person AI lab that became Anthropic and Google's secret weapon | Edwin Chen (Surge AI)","href":"/intel/d6aac06e-b988-4207-8c97-94c17fbe4707","hook":"Surge AI supplies the human data, evals, and RL environments behind leading models, and this conversation explains why benchmark-chasing misleads training an…"}]},{"id":"google/gemini-embedding-2-preview","name":"Gemini Embedding 2 Preview","releaseLabel":"2 Preview","generationId":"2","generationLabel":"2","variantLabel":"Preview","createdAt":"2026-04-17T14:34:25.000Z","kind":"embedding","contextLength":8192,"inputModalities":["text","image","file","audio","video"],"outputModalities":["embeddings"],"promptPerMillion":0.19999999999999998,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Gemini Embedding 2 Preview is Google's first multimodal embedding model. We currently support mapping text and images into a unified vector space for semantic search and retrieval-augmented generation (RAG). It...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"google/gemini-embedding-2-preview","url":"https://openrouter.ai/google/gemini-embedding-2-preview","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"},{"id":"d6aac06e-b988-4207-8c97-94c17fbe4707","title":"The 100-person AI lab that became Anthropic and Google's secret weapon | Edwin Chen (Surge AI)","href":"/intel/d6aac06e-b988-4207-8c97-94c17fbe4707","hook":"Surge AI supplies the human data, evals, and RL environments behind leading models, and this conversation explains why benchmark-chasing misleads training an…"}]},{"id":"google/gemini-embedding-2","name":"Gemini Embedding 2","releaseLabel":"2","generationId":"2","generationLabel":"2","variantLabel":"Base","createdAt":"2026-05-20T15:15:35.000Z","kind":"embedding","contextLength":8192,"inputModalities":["text","image","file","audio","video"],"outputModalities":["embeddings"],"promptPerMillion":0.19999999999999998,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Gemini Embedding 2 is Google's first multimodal embedding model. We currently support mapping text and images into a unified vector space for semantic search and retrieval-augmented generation (RAG). It supports...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"google/gemini-embedding-2","url":"https://openrouter.ai/google/gemini-embedding-2","free":false,"batch":false},{"provider":"OpenRouter","modelId":"google/gemini-embedding-2:batch","url":"https://openrouter.ai/google/gemini-embedding-2:batch","free":false,"batch":true}],"benchmarks":[],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"},{"id":"d6aac06e-b988-4207-8c97-94c17fbe4707","title":"The 100-person AI lab that became Anthropic and Google's secret weapon | Edwin Chen (Surge AI)","href":"/intel/d6aac06e-b988-4207-8c97-94c17fbe4707","hook":"Surge AI supplies the human data, evals, and RL environments behind leading models, and this conversation explains why benchmark-chasing misleads training an…"}]}]},{"id":"veo","label":"Veo","models":[{"id":"google/veo-3.1-20260320","name":"Veo 3.1","releaseLabel":"3.1","generationId":"3.1","generationLabel":"3.1","variantLabel":"Base","createdAt":"2026-03-23T14:45:48.000Z","kind":"video","contextLength":0,"inputModalities":["text","image"],"outputModalities":["video"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Google's state-of-the-art video generation model, built for maximum visual fidelity in final production cuts. Veo 3.1 generates high-quality 1080p video from text or image prompts with native synchronized audio —...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"google/veo-3.1","url":"https://openrouter.ai/google/veo-3.1","free":true,"batch":false}],"benchmarks":[{"benchmark":"AA Video Arena Elo","score":1090,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/video/leaderboard/text-to-video","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Video Arena","url":"https://artificialanalysis.ai/video/leaderboard/text-to-video","note":"Blind preference ranking for video generation"}],"intel":[{"id":"c0fd2d93-7f7e-43b4-aecb-2895139a3fef","title":"#475 – Demis Hassabis: Future of AI, Simulating Reality, Physics and Video Games","href":"/intel/c0fd2d93-7f7e-43b4-aecb-2895139a3fef","hook":"Hassabis lays out DeepMind's concrete views on the path to AGI, scaling law limits, AlphaEvolve, and world-simulation models like Veo 3 — useful context for …"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"}]},{"id":"google/veo-3.1-lite-20260331","name":"Veo 3.1 Lite","releaseLabel":"3.1 Lite","generationId":"3.1","generationLabel":"3.1","variantLabel":"Lite","createdAt":"2026-04-23T21:13:38.000Z","kind":"video","contextLength":0,"inputModalities":["text","image"],"outputModalities":["video"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Google's most cost-effective video generation model, designed for high-volume applications and rapid iteration. Veo 3.1 Lite generates 720p and 1080p video from text or image prompts with native synchronized audio...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"google/veo-3.1-lite","url":"https://openrouter.ai/google/veo-3.1-lite","free":true,"batch":false}],"benchmarks":[{"benchmark":"AA Video Arena Elo","score":1088,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/video/leaderboard/text-to-video","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Video Arena","url":"https://artificialanalysis.ai/video/leaderboard/text-to-video","note":"Blind preference ranking for video generation"}],"intel":[{"id":"c0fd2d93-7f7e-43b4-aecb-2895139a3fef","title":"#475 – Demis Hassabis: Future of AI, Simulating Reality, Physics and Video Games","href":"/intel/c0fd2d93-7f7e-43b4-aecb-2895139a3fef","hook":"Hassabis lays out DeepMind's concrete views on the path to AGI, scaling law limits, AlphaEvolve, and world-simulation models like Veo 3 — useful context for …"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"}]},{"id":"google/veo-3.1-fast-20260320","name":"Veo 3.1 Fast","releaseLabel":"3.1 Fast","generationId":"3.1","generationLabel":"3.1","variantLabel":"Fast","createdAt":"2026-04-24T01:37:46.000Z","kind":"video","contextLength":0,"inputModalities":["text","image"],"outputModalities":["video"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Google's mid-tier video generation model balancing speed and quality. Veo 3.1 Fast generates high-quality video from text or image prompts with native synchronized audio, offering faster turnaround than Veo 3.1...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"google/veo-3.1-fast","url":"https://openrouter.ai/google/veo-3.1-fast","free":true,"batch":false}],"benchmarks":[{"benchmark":"AA Video Arena Elo","score":1086,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/video/leaderboard/text-to-video","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Video Arena","url":"https://artificialanalysis.ai/video/leaderboard/text-to-video","note":"Blind preference ranking for video generation"}],"intel":[{"id":"c0fd2d93-7f7e-43b4-aecb-2895139a3fef","title":"#475 – Demis Hassabis: Future of AI, Simulating Reality, Physics and Video Games","href":"/intel/c0fd2d93-7f7e-43b4-aecb-2895139a3fef","hook":"Hassabis lays out DeepMind's concrete views on the path to AGI, scaling law limits, AlphaEvolve, and world-simulation models like Veo 3 — useful context for …"},{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"}]}]},{"id":"lyria","label":"Lyria","models":[{"id":"google/lyria-3-clip-preview-20260330","name":"Lyria 3 Clip Preview","releaseLabel":"3 Clip Preview","generationId":"3","generationLabel":"3","variantLabel":"Clip Preview","createdAt":"2026-03-30T21:47:35.000Z","kind":"audio","contextLength":1048576,"inputModalities":["text","image"],"outputModalities":["text","audio"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"30 second duration clips are priced at $0.04 per clip. Lyria 3 is Google's family of music generation models, available through the Gemini API. With Lyria 3, you can generate...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"google/lyria-3-clip-preview","url":"https://openrouter.ai/google/lyria-3-clip-preview","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"},{"id":"d6aac06e-b988-4207-8c97-94c17fbe4707","title":"The 100-person AI lab that became Anthropic and Google's secret weapon | Edwin Chen (Surge AI)","href":"/intel/d6aac06e-b988-4207-8c97-94c17fbe4707","hook":"Surge AI supplies the human data, evals, and RL environments behind leading models, and this conversation explains why benchmark-chasing misleads training an…"}]},{"id":"google/lyria-3-pro-preview-20260330","name":"Lyria 3 Pro Preview","releaseLabel":"3 Pro Preview","generationId":"3","generationLabel":"3","variantLabel":"Pro Preview","createdAt":"2026-03-30T21:48:06.000Z","kind":"audio","contextLength":1048576,"inputModalities":["text","image"],"outputModalities":["text","audio"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Full-length songs are priced at $0.08 per song. Lyria 3 is Google's family of music generation models, available through the Gemini API. With Lyria 3, you can generate high-quality, 48kHz...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"google/lyria-3-pro-preview","url":"https://openrouter.ai/google/lyria-3-pro-preview","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"},{"id":"d6aac06e-b988-4207-8c97-94c17fbe4707","title":"The 100-person AI lab that became Anthropic and Google's secret weapon | Edwin Chen (Surge AI)","href":"/intel/d6aac06e-b988-4207-8c97-94c17fbe4707","hook":"Surge AI supplies the human data, evals, and RL environments behind leading models, and this conversation explains why benchmark-chasing misleads training an…"}]}]},{"id":"chirp","label":"Chirp","models":[{"id":"google/chirp-3","name":"Chirp 3","releaseLabel":"3","generationId":"3","generationLabel":"3","variantLabel":"Base","createdAt":"2026-05-05T16:16:23.000Z","kind":"audio","contextLength":0,"inputModalities":["audio"],"outputModalities":["transcription"],"promptPerMillion":16000,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Chirp 3 is Google's latest multilingual speech-to-text model. It offers enhanced transcription accuracy across 24 GA languages and 77+ preview languages, with support for automatic language detection, automatic punctuation, and...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"google/chirp-3","url":"https://openrouter.ai/google/chirp-3","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"},{"id":"d6aac06e-b988-4207-8c97-94c17fbe4707","title":"The 100-person AI lab that became Anthropic and Google's secret weapon | Edwin Chen (Surge AI)","href":"/intel/d6aac06e-b988-4207-8c97-94c17fbe4707","hook":"Surge AI supplies the human data, evals, and RL environments behind leading models, and this conversation explains why benchmark-chasing misleads training an…"}]}]},{"id":"gemini-tts","label":"Gemini TTS","models":[{"id":"google/gemini-3.1-flash-tts-preview","name":"Gemini 3.1 Flash TTS Preview","releaseLabel":"Gemini 3.1 Flash TTS Preview","generationId":"3.1","generationLabel":"3.1","variantLabel":"Flash Preview","createdAt":"2026-04-24T02:55:08.000Z","kind":"audio","contextLength":32768,"inputModalities":["text"],"outputModalities":["speech"],"promptPerMillion":1,"completionPerMillion":20,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Gemini 3.1 Flash TTS Preview is a text-to-speech model from Google, and a substantial generational step up from Gemini 2.5 Flash TTS. It takes text input and produces audio output...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"google/gemini-3.1-flash-tts-preview","url":"https://openrouter.ai/google/gemini-3.1-flash-tts-preview","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[{"id":"9e223b2c-a083-40ee-ac1a-9c1d8f42c981","title":"How to Land a Job at a Frontier Lab","href":"/intel/9e223b2c-a083-40ee-ac1a-9c1d8f42c981","hook":"If you're targeting a job at OpenAI, Anthropic, or Google DeepMind, most advice you'll find online comes from recruiters or people who interviewed there once…"},{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"},{"id":"d6aac06e-b988-4207-8c97-94c17fbe4707","title":"The 100-person AI lab that became Anthropic and Google's secret weapon | Edwin Chen (Surge AI)","href":"/intel/d6aac06e-b988-4207-8c97-94c17fbe4707","hook":"Surge AI supplies the human data, evals, and RL environments behind leading models, and this conversation explains why benchmark-chasing misleads training an…"}]}]}]},{"id":"x-ai","name":"xAI","domain":"x.ai","major":true,"modelCount":12,"series":[{"id":"grok","label":"Grok","models":[{"id":"x-ai/grok-4.20-20260309","name":"Grok 4.20","releaseLabel":"4.20","generationId":"4.20","generationLabel":"4.20","variantLabel":"Base","createdAt":"2026-03-31T17:43:39.000Z","kind":"multimodal","contextLength":2000000,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":1.25,"completionPerMillion":2.5,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Grok 4.20 is a reasoning model from SpaceXAI with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"x-ai/grok-4.20","url":"https://openrouter.ai/x-ai/grok-4.20","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"13536e8a-ee96-4b9d-a315-94221c1fac80","title":"xai-org/grok-build, now open source","href":"/intel/13536e8a-ee96-4b9d-a315-94221c1fac80","hook":"A cautionary tale about what a coding agent can exfiltrate by default — worth reading before you run any new CLI agent in a sensitive directory, and the open…"},{"id":"179adf90-7b00-42fd-8c70-3f909644f45b","title":"The Pulse: Grok’s CLI caught uploading all your local files to the cloud","href":"/intel/179adf90-7b00-42fd-8c70-3f909644f45b","hook":"If you use AI coding CLIs, this breaks down how Grok's CLI was caught silently uploading local files to the cloud — essential reading for understanding the d…"},{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"}]},{"id":"x-ai/grok-4.20-multi-agent-20260309","name":"Grok 4.20 Multi-Agent","releaseLabel":"4.20 Multi-Agent","generationId":"4.20","generationLabel":"4.20","variantLabel":"Multi Agent","createdAt":"2026-03-31T17:45:58.000Z","kind":"multimodal","contextLength":2000000,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":1.25,"completionPerMillion":2.5,"imageOutput":null,"supportsTools":false,"supportsReasoning":true,"description":"Grok 4.20 Multi-Agent is a variant of SpaceXAI’s Grok 4.20 designed for collaborative, agent-based workflows. Multiple agents operate in parallel to conduct deep research, coordinate tool use, and synthesize information...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"x-ai/grok-4.20-multi-agent","url":"https://openrouter.ai/x-ai/grok-4.20-multi-agent","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"13536e8a-ee96-4b9d-a315-94221c1fac80","title":"xai-org/grok-build, now open source","href":"/intel/13536e8a-ee96-4b9d-a315-94221c1fac80","hook":"A cautionary tale about what a coding agent can exfiltrate by default — worth reading before you run any new CLI agent in a sensitive directory, and the open…"},{"id":"179adf90-7b00-42fd-8c70-3f909644f45b","title":"The Pulse: Grok’s CLI caught uploading all your local files to the cloud","href":"/intel/179adf90-7b00-42fd-8c70-3f909644f45b","hook":"If you use AI coding CLIs, this breaks down how Grok's CLI was caught silently uploading local files to the cloud — essential reading for understanding the d…"},{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"}]},{"id":"x-ai/grok-4.3-20260430","name":"Grok 4.3","releaseLabel":"4.3","generationId":"4.3","generationLabel":"4.3","variantLabel":"Base","createdAt":"2026-04-30T23:30:21.000Z","kind":"multimodal","contextLength":1000000,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":1.25,"completionPerMillion":2.5,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Grok 4.3 is a reasoning model from SpaceXAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"x-ai/grok-4.3","url":"https://openrouter.ai/x-ai/grok-4.3","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":63.3,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1397,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 141 · 61,016 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"13536e8a-ee96-4b9d-a315-94221c1fac80","title":"xai-org/grok-build, now open source","href":"/intel/13536e8a-ee96-4b9d-a315-94221c1fac80","hook":"A cautionary tale about what a coding agent can exfiltrate by default — worth reading before you run any new CLI agent in a sensitive directory, and the open…"},{"id":"179adf90-7b00-42fd-8c70-3f909644f45b","title":"The Pulse: Grok’s CLI caught uploading all your local files to the cloud","href":"/intel/179adf90-7b00-42fd-8c70-3f909644f45b","hook":"If you use AI coding CLIs, this breaks down how Grok's CLI was caught silently uploading local files to the cloud — essential reading for understanding the d…"},{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"}]},{"id":"x-ai/grok-4.5-20260708","name":"Grok 4.5","releaseLabel":"4.5","generationId":"4.5","generationLabel":"4.5","variantLabel":"Base","createdAt":"2026-07-08T15:05:54.000Z","kind":"multimodal","contextLength":500000,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":2,"completionPerMillion":6,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Grok 4.5 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM.","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"x-ai/grok-4.5","url":"https://openrouter.ai/x-ai/grok-4.5","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":77.1,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1452,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 37 · 22,030 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"13536e8a-ee96-4b9d-a315-94221c1fac80","title":"xai-org/grok-build, now open source","href":"/intel/13536e8a-ee96-4b9d-a315-94221c1fac80","hook":"A cautionary tale about what a coding agent can exfiltrate by default — worth reading before you run any new CLI agent in a sensitive directory, and the open…"},{"id":"179adf90-7b00-42fd-8c70-3f909644f45b","title":"The Pulse: Grok’s CLI caught uploading all your local files to the cloud","href":"/intel/179adf90-7b00-42fd-8c70-3f909644f45b","hook":"If you use AI coding CLIs, this breaks down how Grok's CLI was caught silently uploading local files to the cloud — essential reading for understanding the d…"},{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"}]},{"id":"x-ai/grok-4.6-20260810","name":"Grok 4.6","releaseLabel":"4.6","generationId":"4.6","generationLabel":"4.6","variantLabel":"Base","createdAt":"2026-08-12T15:35:57.000Z","kind":"multimodal","contextLength":500000,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":2,"completionPerMillion":6,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Grok 4.6 is SpaceXAI's smartest model with frontier performance on coding, knowledge work, and STEM.","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"x-ai/grok-4.6","url":"https://openrouter.ai/x-ai/grok-4.6","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":79,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"AA Intelligence Index","score":60.9,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/grok-4-6","measuredAt":"2026-08-27","scope":"Current public index; highest tested effort configuration"},{"benchmark":"LMArena overall","score":1444,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 54 · 3,473 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"13536e8a-ee96-4b9d-a315-94221c1fac80","title":"xai-org/grok-build, now open source","href":"/intel/13536e8a-ee96-4b9d-a315-94221c1fac80","hook":"A cautionary tale about what a coding agent can exfiltrate by default — worth reading before you run any new CLI agent in a sensitive directory, and the open…"},{"id":"179adf90-7b00-42fd-8c70-3f909644f45b","title":"The Pulse: Grok’s CLI caught uploading all your local files to the cloud","href":"/intel/179adf90-7b00-42fd-8c70-3f909644f45b","hook":"If you use AI coding CLIs, this breaks down how Grok's CLI was caught silently uploading local files to the cloud — essential reading for understanding the d…"},{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"}]}]},{"id":"grok-imagine-image","label":"Grok Imagine Image","models":[{"id":"x-ai/grok-imagine-image-quality-20260512","name":"Grok Imagine Image Quality","releaseLabel":"Quality","generationId":"initial","generationLabel":"Initial","variantLabel":"Quality","createdAt":"2026-05-18T15:19:44.000Z","kind":"image","contextLength":65536,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":11.9760479041916,"supportsTools":false,"supportsReasoning":false,"description":"Grok Imagine Image Quality is SpaceXAI's fast, high-fidelity image generation and editing model. It accepts text prompts and optional reference images, producing photorealistic outputs at 1K or 2K across a...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"x-ai/grok-imagine-image-quality","url":"https://openrouter.ai/x-ai/grok-imagine-image-quality","free":true,"batch":false}],"benchmarks":[{"benchmark":"AA Image Arena Elo","score":1236,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/image/leaderboard/text-to-image","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"},{"id":"13536e8a-ee96-4b9d-a315-94221c1fac80","title":"xai-org/grok-build, now open source","href":"/intel/13536e8a-ee96-4b9d-a315-94221c1fac80","hook":"A cautionary tale about what a coding agent can exfiltrate by default — worth reading before you run any new CLI agent in a sensitive directory, and the open…"}]},{"id":"x-ai/grok-imagine-image-2.0-20260811","name":"Grok Imagine Image 2.0","releaseLabel":"2.0","generationId":"2.0","generationLabel":"2.0","variantLabel":"Base","createdAt":"2026-08-11T22:07:24.000Z","kind":"image","contextLength":65536,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":9.580838323353289,"supportsTools":false,"supportsReasoning":false,"description":"Grok Imagine Image 2.0 is an image generation and editing model from xAI. It is suited for creating images from text prompts and editing images from references, with low and...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"x-ai/grok-imagine-image-2.0","url":"https://openrouter.ai/x-ai/grok-imagine-image-2.0","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"},{"id":"13536e8a-ee96-4b9d-a315-94221c1fac80","title":"xai-org/grok-build, now open source","href":"/intel/13536e8a-ee96-4b9d-a315-94221c1fac80","hook":"A cautionary tale about what a coding agent can exfiltrate by default — worth reading before you run any new CLI agent in a sensitive directory, and the open…"}]}]},{"id":"grok-imagine-video","label":"Grok Imagine Video","models":[{"id":"x-ai/grok-imagine-video-20260512","name":"Grok Imagine Video","releaseLabel":"Grok Imagine Video","generationId":"initial","generationLabel":"Initial","variantLabel":"Base","createdAt":"2026-05-18T15:19:46.000Z","kind":"video","contextLength":0,"inputModalities":["text","image"],"outputModalities":["video"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Grok Imagine Video is SpaceXAI's fast, text-, image-, and reference-conditioned video generation model. It produces short videos (1–15 seconds, 24 fps) at 480p or 720p across seven aspect ratios -...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"x-ai/grok-imagine-video","url":"https://openrouter.ai/x-ai/grok-imagine-video","free":true,"batch":false}],"benchmarks":[{"benchmark":"AA Video Arena Elo","score":1062,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/video/leaderboard/text-to-video","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Video Arena","url":"https://artificialanalysis.ai/video/leaderboard/text-to-video","note":"Blind preference ranking for video generation"}],"intel":[{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"},{"id":"13536e8a-ee96-4b9d-a315-94221c1fac80","title":"xai-org/grok-build, now open source","href":"/intel/13536e8a-ee96-4b9d-a315-94221c1fac80","hook":"A cautionary tale about what a coding agent can exfiltrate by default — worth reading before you run any new CLI agent in a sensitive directory, and the open…"}]},{"id":"x-ai/grok-imagine-video-1.5-20260719","name":"Grok Imagine Video 1.5","releaseLabel":"1.5","generationId":"1.5","generationLabel":"1.5","variantLabel":"Base","createdAt":"2026-07-20T11:48:20.000Z","kind":"video","contextLength":0,"inputModalities":["text","image"],"outputModalities":["video"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Grok Imagine Video 1.5 is a video generation model from SpaceXAI. It creates videos from text prompts, with an optional starting image to guide the scene. It can direct subject...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"x-ai/grok-imagine-video-1.5","url":"https://openrouter.ai/x-ai/grok-imagine-video-1.5","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Video Arena","url":"https://artificialanalysis.ai/video/leaderboard/text-to-video","note":"Blind preference ranking for video generation"}],"intel":[{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"},{"id":"13536e8a-ee96-4b9d-a315-94221c1fac80","title":"xai-org/grok-build, now open source","href":"/intel/13536e8a-ee96-4b9d-a315-94221c1fac80","hook":"A cautionary tale about what a coding agent can exfiltrate by default — worth reading before you run any new CLI agent in a sensitive directory, and the open…"}]}]},{"id":"grok-build","label":"Grok Build","models":[{"id":"x-ai/grok-build-0.1-20260520","name":"Grok Build 0.1","releaseLabel":"0.1","generationId":"0.1","generationLabel":"0.1","variantLabel":"Base","createdAt":"2026-05-20T17:28:43.000Z","kind":"multimodal","contextLength":256000,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":1,"completionPerMillion":2,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Grok Build 0.1 is SpaceXAI’s fast coding model trained specifically for agentic software engineering workflows. It supports text and image inputs with text output, and is optimized for interactive coding...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"x-ai/grok-build-0.1","url":"https://openrouter.ai/x-ai/grok-build-0.1","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":68.6,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"},{"id":"13536e8a-ee96-4b9d-a315-94221c1fac80","title":"xai-org/grok-build, now open source","href":"/intel/13536e8a-ee96-4b9d-a315-94221c1fac80","hook":"A cautionary tale about what a coding agent can exfiltrate by default — worth reading before you run any new CLI agent in a sensitive directory, and the open…"}]}]},{"id":"grok-transcribe","label":"Grok Transcribe","models":[{"id":"x-ai/grok-stt-20260723","name":"Grok STT 1.0","releaseLabel":"Grok STT 1.0","generationId":"1.0","generationLabel":"1.0","variantLabel":"1.0","createdAt":"2026-07-23T14:30:14.000Z","kind":"audio","contextLength":0,"inputModalities":["audio"],"outputModalities":["transcription"],"promptPerMillion":100000,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Grok STT is SpaceXAI's speech-to-text model, available via the REST /v1/stt endpoint. It supports transcription with word-level timestamps, optional speaker diarization, and multichannel audio.","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"x-ai/grok-stt-1.0","url":"https://openrouter.ai/x-ai/grok-stt-1.0","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"},{"id":"13536e8a-ee96-4b9d-a315-94221c1fac80","title":"xai-org/grok-build, now open source","href":"/intel/13536e8a-ee96-4b9d-a315-94221c1fac80","hook":"A cautionary tale about what a coding agent can exfiltrate by default — worth reading before you run any new CLI agent in a sensitive directory, and the open…"}]}]},{"id":"grok-voice","label":"Grok Voice","models":[{"id":"x-ai/grok-voice-tts-1.0","name":"Grok Voice TTS 1.0","releaseLabel":"TTS 1.0","generationId":"1.0","generationLabel":"1.0","variantLabel":"TTS","createdAt":"2026-05-15T00:37:36.000Z","kind":"audio","contextLength":15000,"inputModalities":["text"],"outputModalities":["speech"],"promptPerMillion":15,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Grok Voice TTS 1.0 is a text-to-speech model from SpaceXAI. It converts text into spoken audio across 20+ languages with automatic language detection, and offers five built-in voices (Eve, Ara,...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"x-ai/grok-voice-tts-1.0","url":"https://openrouter.ai/x-ai/grok-voice-tts-1.0","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"},{"id":"13536e8a-ee96-4b9d-a315-94221c1fac80","title":"xai-org/grok-build, now open source","href":"/intel/13536e8a-ee96-4b9d-a315-94221c1fac80","hook":"A cautionary tale about what a coding agent can exfiltrate by default — worth reading before you run any new CLI agent in a sensitive directory, and the open…"}]}]}]},{"id":"deepseek","name":"DeepSeek","domain":"deepseek.com","major":true,"modelCount":14,"series":[{"id":"deepseek","label":"DeepSeek","models":[{"id":"deepseek/deepseek-chat-v3","name":"DeepSeek V3","releaseLabel":"V3","generationId":"releases","generationLabel":"Releases","variantLabel":"V3","createdAt":"2024-12-26T19:28:40.000Z","kind":"language","contextLength":163840,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.2574,"completionPerMillion":1.0287,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"DeepSeek-V3 is the latest model from the DeepSeek team, building upon the instruction following and coding abilities of the previous versions. Pre-trained on nearly 15 trillion tokens, the reported evaluations...","huggingFaceId":"deepseek-ai/DeepSeek-V3","routes":[{"provider":"OpenRouter","modelId":"deepseek/deepseek-chat","url":"https://openrouter.ai/deepseek/deepseek-chat","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":61.2,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2024_11_25.csv","measuredAt":"2024-11-25","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1333,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 203 · 21,770 votes"},{"benchmark":"GSM8K","score":89.3,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"MMLU-Pro","score":64.4,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"signalbench · src","score":0.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation · signalbench raw per-item responses"},{"benchmark":"signalbench · time","score":0.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation"},{"benchmark":"signalbench · access deny","score":0.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation"},{"benchmark":"signalbench · memory label","score":0.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation"},{"benchmark":"signalbench · injection","score":0.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation"},{"benchmark":"signalbench · bot policy","score":0.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation"},{"benchmark":"gpqa · diamond","score":58.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/0a/db/0adbcd9f-38e6-414a-88b1-7bd928d4c72e.json","measuredAt":"2025-03-19","scope":"Registered model evaluation · EvalEval"},{"benchmark":"LEXam · open question","score":52.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://lexam-benchmark.github.io/","measuredAt":"2026-06-02","scope":"Registered model evaluation · LEXam Leaderboard"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"069c108b-a77f-4cce-bf04-28d58c2f91c1","title":"The State Of LLMs 2025","href":"/intel/069c108b-a77f-4cce-bf04-28d58c2f91c1","hook":"It distills a chaotic year of LLM research — DeepSeek's cost disruption, the rise of RLVR/GRPO for reasoning, and inference-time scaling — into one technical…"},{"id":"4425150c-cacb-4f7f-9b74-ec7df018ff9d","title":"A Technical Tour of the DeepSeek Models from V3 to V3.2","href":"/intel/4425150c-cacb-4f7f-9b74-ec7df018ff9d","hook":"If you're evaluating open-weight LLMs for agentic work, this walks through DeepSeek's architectural evolution from V3 to V3.2 — including the DeepSeek Sparse…"},{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"}]},{"id":"deepseek/deepseek-chat-v3-0324","name":"DeepSeek V3 0324","releaseLabel":"V3 0324","generationId":"v3","generationLabel":"V3","variantLabel":"0324","createdAt":"2025-03-24T13:59:15.000Z","kind":"language","contextLength":163840,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.25,"completionPerMillion":1,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"DeepSeek V3, a 685B-parameter, mixture-of-experts model, is the latest iteration of the flagship chat model family from the DeepSeek team. It succeeds the [DeepSeek V3](/deepseek/deepseek-chat-v3) model and performs really well...","huggingFaceId":"deepseek-ai/DeepSeek-V3-0324","routes":[{"provider":"OpenRouter","modelId":"deepseek/deepseek-chat-v3-0324","url":"https://openrouter.ai/deepseek/deepseek-chat-v3-0324","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":63.5,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2025_04_25.csv","measuredAt":"2025-04-25","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1375,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 159 · 45,595 votes"},{"benchmark":"MMLU-Pro","score":81.2,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"MMLU-Pro · mmlu pro","score":81.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V3-0324","measuredAt":"2026-01-28","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"069c108b-a77f-4cce-bf04-28d58c2f91c1","title":"The State Of LLMs 2025","href":"/intel/069c108b-a77f-4cce-bf04-28d58c2f91c1","hook":"It distills a chaotic year of LLM research — DeepSeek's cost disruption, the rise of RLVR/GRPO for reasoning, and inference-time scaling — into one technical…"},{"id":"4425150c-cacb-4f7f-9b74-ec7df018ff9d","title":"A Technical Tour of the DeepSeek Models from V3 to V3.2","href":"/intel/4425150c-cacb-4f7f-9b74-ec7df018ff9d","hook":"If you're evaluating open-weight LLMs for agentic work, this walks through DeepSeek's architectural evolution from V3 to V3.2 — including the DeepSeek Sparse…"},{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"}]},{"id":"deepseek/deepseek-chat-v3.1","name":"DeepSeek V3.1","releaseLabel":"V3.1","generationId":"v3.1","generationLabel":"V3.1","variantLabel":"Base","createdAt":"2025-08-21T12:33:48.000Z","kind":"language","contextLength":163840,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.55,"completionPerMillion":1.6500000000000001,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...","huggingFaceId":"deepseek-ai/DeepSeek-V3.1","routes":[{"provider":"OpenRouter","modelId":"deepseek/deepseek-chat-v3.1","url":"https://openrouter.ai/deepseek/deepseek-chat-v3.1","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1419,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 105 · 15,002 votes"},{"benchmark":"MMLU-Pro · mmlu pro","score":84.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/cd/06/cd069a5d-dff4-4472-9dfa-8b275c5483f0.json","measuredAt":"2026-06-30","scope":"Registered model evaluation · EvalEval"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"069c108b-a77f-4cce-bf04-28d58c2f91c1","title":"The State Of LLMs 2025","href":"/intel/069c108b-a77f-4cce-bf04-28d58c2f91c1","hook":"It distills a chaotic year of LLM research — DeepSeek's cost disruption, the rise of RLVR/GRPO for reasoning, and inference-time scaling — into one technical…"},{"id":"4425150c-cacb-4f7f-9b74-ec7df018ff9d","title":"A Technical Tour of the DeepSeek Models from V3 to V3.2","href":"/intel/4425150c-cacb-4f7f-9b74-ec7df018ff9d","hook":"If you're evaluating open-weight LLMs for agentic work, this walks through DeepSeek's architectural evolution from V3 to V3.2 — including the DeepSeek Sparse…"},{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"}]},{"id":"deepseek/deepseek-v3.1-terminus","name":"DeepSeek V3.1 Terminus","releaseLabel":"V3.1 Terminus","generationId":"v3.1","generationLabel":"V3.1","variantLabel":"Terminus","createdAt":"2025-09-22T13:37:55.000Z","kind":"language","contextLength":163840,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.27,"completionPerMillion":1,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"DeepSeek-V3.1 Terminus is an update to [DeepSeek V3.1](/deepseek/deepseek-chat-v3.1) that maintains the model's original capabilities while addressing issues reported by users, including language consistency and agent capabilities, further optimizing the model's...","huggingFaceId":"deepseek-ai/DeepSeek-V3.1-Terminus","routes":[{"provider":"OpenRouter","modelId":"deepseek/deepseek-v3.1-terminus","url":"https://openrouter.ai/deepseek/deepseek-v3.1-terminus","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":71.9,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2025_05_30.csv","measuredAt":"2025-05-30","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1420,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 102 · 3,470 votes"},{"benchmark":"gpqa · diamond","score":74.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/f3/b0/f3b076c8-a51d-4c88-b86e-652f7c74f5b3.json","measuredAt":"2026-04-17","scope":"Registered model evaluation · EvalEval"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"069c108b-a77f-4cce-bf04-28d58c2f91c1","title":"The State Of LLMs 2025","href":"/intel/069c108b-a77f-4cce-bf04-28d58c2f91c1","hook":"It distills a chaotic year of LLM research — DeepSeek's cost disruption, the rise of RLVR/GRPO for reasoning, and inference-time scaling — into one technical…"},{"id":"4425150c-cacb-4f7f-9b74-ec7df018ff9d","title":"A Technical Tour of the DeepSeek Models from V3 to V3.2","href":"/intel/4425150c-cacb-4f7f-9b74-ec7df018ff9d","hook":"If you're evaluating open-weight LLMs for agentic work, this walks through DeepSeek's architectural evolution from V3 to V3.2 — including the DeepSeek Sparse…"},{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"}]},{"id":"deepseek/deepseek-v3.2-exp","name":"DeepSeek V3.2 Exp","releaseLabel":"V3.2 Exp","generationId":"v3.2","generationLabel":"V3.2","variantLabel":"Exp","createdAt":"2025-09-29T12:54:41.000Z","kind":"language","contextLength":163840,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.27,"completionPerMillion":0.41,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"DeepSeek-V3.2-Exp is an experimental large language model released by DeepSeek as an intermediate step between V3.1 and future architectures. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","huggingFaceId":"deepseek-ai/DeepSeek-V3.2-Exp","routes":[{"provider":"OpenRouter","modelId":"deepseek/deepseek-v3.2-exp","url":"https://openrouter.ai/deepseek/deepseek-v3.2-exp","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":58.9,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1424,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 91 · 9,109 votes"},{"benchmark":"LEXam · open question","score":56.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://lexam-benchmark.github.io/","measuredAt":"2026-06-02","scope":"Registered model evaluation · LEXam Leaderboard"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"069c108b-a77f-4cce-bf04-28d58c2f91c1","title":"The State Of LLMs 2025","href":"/intel/069c108b-a77f-4cce-bf04-28d58c2f91c1","hook":"It distills a chaotic year of LLM research — DeepSeek's cost disruption, the rise of RLVR/GRPO for reasoning, and inference-time scaling — into one technical…"},{"id":"4425150c-cacb-4f7f-9b74-ec7df018ff9d","title":"A Technical Tour of the DeepSeek Models from V3 to V3.2","href":"/intel/4425150c-cacb-4f7f-9b74-ec7df018ff9d","hook":"If you're evaluating open-weight LLMs for agentic work, this walks through DeepSeek's architectural evolution from V3 to V3.2 — including the DeepSeek Sparse…"},{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"}]},{"id":"deepseek/deepseek-v3.2-20251201","name":"DeepSeek V3.2","releaseLabel":"V3.2","generationId":"v3.2","generationLabel":"V3.2","variantLabel":"Base","createdAt":"2025-12-01T13:10:42.000Z","kind":"language","contextLength":163840,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.26,"completionPerMillion":0.38,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","huggingFaceId":"deepseek-ai/DeepSeek-V3.2","routes":[{"provider":"OpenRouter","modelId":"deepseek/deepseek-v3.2","url":"https://openrouter.ai/deepseek/deepseek-v3.2","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":63.1,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1425,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 90 · 47,295 votes"},{"benchmark":"AIME 2026","score":94.2,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"EvasionBench","score":66.9,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"GPQA","score":82.4,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"Humanity's Last Exam","score":40.8,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"HMMT 2026","score":84.1,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"MMLU-Pro","score":85,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"SWE-bench Pro","score":15.6,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"SWE-bench Verified","score":70,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"Terminal-Bench","score":39.6,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"WildClawBench · overall","score":34,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://internlm.github.io/WildClawBench","measuredAt":"2026-05-22","scope":"Registered model evaluation · WildClawBench"},{"benchmark":"results · overall","score":0.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://www.exgentic.ai","measuredAt":"2026-05-18","scope":"Registered model evaluation · Open Agent Leaderboard"},{"benchmark":"results · appworld","score":0,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://www.exgentic.ai","measuredAt":"2026-05-18","scope":"Registered model evaluation · Open Agent Leaderboard"},{"benchmark":"results · browsecomp plus","score":0.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://www.exgentic.ai","measuredAt":"2026-05-18","scope":"Registered model evaluation · Open Agent Leaderboard"},{"benchmark":"results · swebench","score":0.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://www.exgentic.ai","measuredAt":"2026-05-18","scope":"Registered model evaluation · Open Agent Leaderboard"},{"benchmark":"results · taubench airline","score":0.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://www.exgentic.ai","measuredAt":"2026-05-18","scope":"Registered model evaluation · Open Agent Leaderboard"},{"benchmark":"results · taubench retail","score":0.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://www.exgentic.ai","measuredAt":"2026-05-18","scope":"Registered model evaluation · Open Agent Leaderboard"},{"benchmark":"results · taubench telecom","score":0.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://www.exgentic.ai","measuredAt":"2026-05-18","scope":"Registered model evaluation · Open Agent Leaderboard"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"069c108b-a77f-4cce-bf04-28d58c2f91c1","title":"The State Of LLMs 2025","href":"/intel/069c108b-a77f-4cce-bf04-28d58c2f91c1","hook":"It distills a chaotic year of LLM research — DeepSeek's cost disruption, the rise of RLVR/GRPO for reasoning, and inference-time scaling — into one technical…"},{"id":"4425150c-cacb-4f7f-9b74-ec7df018ff9d","title":"A Technical Tour of the DeepSeek Models from V3 to V3.2","href":"/intel/4425150c-cacb-4f7f-9b74-ec7df018ff9d","hook":"If you're evaluating open-weight LLMs for agentic work, this walks through DeepSeek's architectural evolution from V3 to V3.2 — including the DeepSeek Sparse…"},{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"}]},{"id":"deepseek/deepseek-v4-flash-20260423","name":"DeepSeek V4 Flash 0423","releaseLabel":"V4 Flash 0423","generationId":"v4","generationLabel":"V4","variantLabel":"Flash","createdAt":"2026-04-24T03:17:46.000Z","kind":"language","contextLength":1048576,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.07644000000000001,"completionPerMillion":0.15288000000000002,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","huggingFaceId":"deepseek-ai/DeepSeek-V4-Flash","routes":[{"provider":"OpenRouter","modelId":"deepseek/deepseek-v4-flash","url":"https://openrouter.ai/deepseek/deepseek-v4-flash","free":false,"batch":false}],"benchmarks":[{"benchmark":"Long-Horizon-Terminal-Bench · lhtb solved","score":2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/IntelligenceLab/LHTB-leaderboard/tree/main/submissions/long-horizon-terminal-bench/1.0-extended/terminus-2__deepseek-v4-flash-3h","measuredAt":"2026-08-08","scope":"Registered model evaluation · LHTB run artifacts"},{"benchmark":"SWE-bench_Multilingual · swe bench multilingual % resolved","score":73.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash","measuredAt":"2026-08-10","scope":"Registered model evaluation · Model Card"},{"benchmark":"skillsbench · skillsbench v1 1","score":44.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/benchflow/skillsbench-leaderboard/raw/main/leaderboard/skillsbench/v1.1/official.json","measuredAt":"2026-06-11","scope":"Registered model evaluation · SkillsBench v1.1 official leaderboard"},{"benchmark":"Claw-Eval · general","score":57.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://claw-eval.github.io","measuredAt":"2026-04-23","scope":"Registered model evaluation · Claw-Eval Leaderboard"},{"benchmark":"Claw-Eval · multi turn","score":57.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://claw-eval.github.io","measuredAt":"2026-04-23","scope":"Registered model evaluation · Claw-Eval Leaderboard"},{"benchmark":"gpqa · diamond","score":88.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash","measuredAt":"2026-04-24","scope":"Registered model evaluation · Model Card"},{"benchmark":"hle · hle","score":34.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash","measuredAt":"2026-04-24","scope":"Registered model evaluation · Model Card"},{"benchmark":"MMLU-Pro · mmlu pro","score":86.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash","measuredAt":"2026-04-24","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"069c108b-a77f-4cce-bf04-28d58c2f91c1","title":"The State Of LLMs 2025","href":"/intel/069c108b-a77f-4cce-bf04-28d58c2f91c1","hook":"It distills a chaotic year of LLM research — DeepSeek's cost disruption, the rise of RLVR/GRPO for reasoning, and inference-time scaling — into one technical…"},{"id":"4425150c-cacb-4f7f-9b74-ec7df018ff9d","title":"A Technical Tour of the DeepSeek Models from V3 to V3.2","href":"/intel/4425150c-cacb-4f7f-9b74-ec7df018ff9d","hook":"If you're evaluating open-weight LLMs for agentic work, this walks through DeepSeek's architectural evolution from V3 to V3.2 — including the DeepSeek Sparse…"},{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"}]},{"id":"deepseek/deepseek-v4-pro-20260423","name":"DeepSeek V4 Pro 0423","releaseLabel":"V4 Pro 0423","generationId":"v4","generationLabel":"V4","variantLabel":"Pro","createdAt":"2026-04-24T03:17:59.000Z","kind":"language","contextLength":1048576,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.87,"completionPerMillion":1.74,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...","huggingFaceId":"deepseek-ai/DeepSeek-V4-Pro","routes":[{"provider":"OpenRouter","modelId":"deepseek/deepseek-v4-pro","url":"https://openrouter.ai/deepseek/deepseek-v4-pro","free":false,"batch":false}],"benchmarks":[{"benchmark":"SWE-bench_Multilingual · swe bench multilingual % resolved","score":76.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro","measuredAt":"2026-08-10","scope":"Registered model evaluation · Model Card"},{"benchmark":"Long-Horizon-Terminal-Bench · lhtb solved","score":3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://zli12321.github.io/LHTB/leaderboard.html","measuredAt":"2026-07-16","scope":"Registered model evaluation · LHTB leaderboard"},{"benchmark":"skillsbench · skillsbench v1 1","score":50.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/benchflow/skillsbench-leaderboard/raw/main/leaderboard/skillsbench/v1.1/official.json","measuredAt":"2026-06-11","scope":"Registered model evaluation · SkillsBench v1.1 official leaderboard"},{"benchmark":"chi-bench · chi bench","score":14.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://arxiv.org/abs/2605.16679","measuredAt":"2026-05-08","scope":"Registered model evaluation · CHI-Bench"},{"benchmark":"chi-bench · prior authorization","score":10.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://arxiv.org/abs/2605.16679","measuredAt":"2026-05-08","scope":"Registered model evaluation · CHI-Bench"},{"benchmark":"chi-bench · utilization management","score":28,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://arxiv.org/abs/2605.16679","measuredAt":"2026-05-08","scope":"Registered model evaluation · CHI-Bench"},{"benchmark":"chi-bench · care management","score":4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://arxiv.org/abs/2605.16679","measuredAt":"2026-05-08","scope":"Registered model evaluation · CHI-Bench"},{"benchmark":"WildClawBench · overall","score":43.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://internlm.github.io/WildClawBench","measuredAt":"2026-05-22","scope":"Registered model evaluation · WildClawBench"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"069c108b-a77f-4cce-bf04-28d58c2f91c1","title":"The State Of LLMs 2025","href":"/intel/069c108b-a77f-4cce-bf04-28d58c2f91c1","hook":"It distills a chaotic year of LLM research — DeepSeek's cost disruption, the rise of RLVR/GRPO for reasoning, and inference-time scaling — into one technical…"},{"id":"4425150c-cacb-4f7f-9b74-ec7df018ff9d","title":"A Technical Tour of the DeepSeek Models from V3 to V3.2","href":"/intel/4425150c-cacb-4f7f-9b74-ec7df018ff9d","hook":"If you're evaluating open-weight LLMs for agentic work, this walks through DeepSeek's architectural evolution from V3 to V3.2 — including the DeepSeek Sparse…"},{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"}]},{"id":"deepseek/deepseek-v4-flash-20260731","name":"DeepSeek V4 Flash 0731","releaseLabel":"V4 Flash 0731","generationId":"v4","generationLabel":"V4","variantLabel":"Flash 0731","createdAt":"2026-07-31T06:21:48.000Z","kind":"language","contextLength":1310720,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.049999999999999996,"completionPerMillion":0.09999999999999999,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....","huggingFaceId":"deepseek-ai/DeepSeek-V4-Flash-0731","routes":[{"provider":"OpenRouter","modelId":"deepseek/deepseek-v4-flash-0731","url":"https://openrouter.ai/deepseek/deepseek-v4-flash-0731","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":74.8,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"terminal-bench-2.1 · terminalbench 2 1","score":82.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-0731","measuredAt":"2026-08-01","scope":"Registered model evaluation · Model Card"},{"benchmark":"deep-swe · deep swe","score":54.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-0731","measuredAt":"2026-08-03","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"069c108b-a77f-4cce-bf04-28d58c2f91c1","title":"The State Of LLMs 2025","href":"/intel/069c108b-a77f-4cce-bf04-28d58c2f91c1","hook":"It distills a chaotic year of LLM research — DeepSeek's cost disruption, the rise of RLVR/GRPO for reasoning, and inference-time scaling — into one technical…"},{"id":"4425150c-cacb-4f7f-9b74-ec7df018ff9d","title":"A Technical Tour of the DeepSeek Models from V3 to V3.2","href":"/intel/4425150c-cacb-4f7f-9b74-ec7df018ff9d","hook":"If you're evaluating open-weight LLMs for agentic work, this walks through DeepSeek's architectural evolution from V3 to V3.2 — including the DeepSeek Sparse…"},{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"}]},{"id":"deepseek/deepseek-v4-pro-20260813","name":"DeepSeek V4 Pro 0813","releaseLabel":"V4 Pro 0813","generationId":"v4","generationLabel":"V4","variantLabel":"Pro 0813","createdAt":"2026-08-12T15:42:44.000Z","kind":"language","contextLength":1048576,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":1.122,"completionPerMillion":3.366,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.","huggingFaceId":"deepseek-ai/DeepSeek-V4-Pro-0813","routes":[{"provider":"OpenRouter","modelId":"deepseek/deepseek-v4-pro-0813","url":"https://openrouter.ai/deepseek/deepseek-v4-pro-0813","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":78.2,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"AA Intelligence Index","score":53.2,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/deepseek-v4-pro","measuredAt":"2026-08-27","scope":"Current public index; highest tested effort configuration"},{"benchmark":"AA Agentic Index","score":49.6,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/deepseek-v4-pro","measuredAt":"2026-08-27","scope":"Current public agentic composite"},{"benchmark":"SciCode","score":49.2,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/deepseek-v4-pro","measuredAt":"2026-08-27","scope":"Scientific coding evaluation"},{"benchmark":"AA Omniscience Index","score":0.8,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/deepseek-v4-pro","measuredAt":"2026-08-27","scope":"Current public knowledge and hallucination index"},{"benchmark":"MLCR overall","score":17.8,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/deepseek-v4-pro","measuredAt":"2026-08-27","scope":"Machine-learning code reasoning evaluation"},{"benchmark":"GDPval-AA normalized","score":53.9,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/deepseek-v4-pro","measuredAt":"2026-08-27","scope":"Normalized real-world knowledge-work evaluation"},{"benchmark":"τ²-bench Banking","score":39.6,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/deepseek-v4-pro","measuredAt":"2026-08-27","scope":"Tool-agent benchmark in the banking domain"},{"benchmark":"Terminal-Bench 2.1","score":78.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/deepseek-v4-pro","measuredAt":"2026-08-27","scope":"Terminal-agent task completion"},{"benchmark":"LCR","score":75.3,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/deepseek-v4-pro","measuredAt":"2026-08-27","scope":"Long-context reasoning evaluation"},{"benchmark":"Humanity's Last Exam","score":41,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/deepseek-v4-pro","measuredAt":"2026-08-27","scope":"Broad expert-level knowledge evaluation"},{"benchmark":"GPQA","score":92.8,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/deepseek-v4-pro","measuredAt":"2026-08-27","scope":"Graduate-level science questions"},{"benchmark":"CritPt","score":18,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/deepseek-v4-pro","measuredAt":"2026-08-27","scope":"Critical-point reasoning evaluation"},{"benchmark":"Enterprise Ops Gym","score":49.6,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/deepseek-v4-pro","measuredAt":"2026-08-27","scope":"Enterprise operations agent tasks"},{"benchmark":"terminal-bench-2.1 · terminalbench 2 1","score":87.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813","measuredAt":"2026-08-13","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"069c108b-a77f-4cce-bf04-28d58c2f91c1","title":"The State Of LLMs 2025","href":"/intel/069c108b-a77f-4cce-bf04-28d58c2f91c1","hook":"It distills a chaotic year of LLM research — DeepSeek's cost disruption, the rise of RLVR/GRPO for reasoning, and inference-time scaling — into one technical…"},{"id":"4425150c-cacb-4f7f-9b74-ec7df018ff9d","title":"A Technical Tour of the DeepSeek Models from V3 to V3.2","href":"/intel/4425150c-cacb-4f7f-9b74-ec7df018ff9d","hook":"If you're evaluating open-weight LLMs for agentic work, this walks through DeepSeek's architectural evolution from V3 to V3.2 — including the DeepSeek Sparse…"},{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"}]},{"id":"deepseek/deepseek-v4-flash-vision-exp-20260821","name":"DeepSeek V4 Flash Vision Exp","releaseLabel":"V4 Flash Vision Exp","generationId":"v4","generationLabel":"V4","variantLabel":"Flash Vision Exp","createdAt":"2026-08-21T11:26:03.000Z","kind":"multimodal","contextLength":1048576,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0.22,"completionPerMillion":0.66,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"DeepSeek V4 Flash Vision Exp is an experimental vision-enabled version of [DeepSeek V4 Flash 0731](https://openrouter.ai/deepseek/deepseek-v4-flash-0731) from DeepSeek, adding image understanding while matching the base model on text capabilities including agents,...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"deepseek/deepseek-v4-flash-vision-exp","url":"https://openrouter.ai/deepseek/deepseek-v4-flash-vision-exp","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":77.7,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"069c108b-a77f-4cce-bf04-28d58c2f91c1","title":"The State Of LLMs 2025","href":"/intel/069c108b-a77f-4cce-bf04-28d58c2f91c1","hook":"It distills a chaotic year of LLM research — DeepSeek's cost disruption, the rise of RLVR/GRPO for reasoning, and inference-time scaling — into one technical…"},{"id":"4425150c-cacb-4f7f-9b74-ec7df018ff9d","title":"A Technical Tour of the DeepSeek Models from V3 to V3.2","href":"/intel/4425150c-cacb-4f7f-9b74-ec7df018ff9d","hook":"If you're evaluating open-weight LLMs for agentic work, this walks through DeepSeek's architectural evolution from V3 to V3.2 — including the DeepSeek Sparse…"},{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"}]}]},{"id":"deepseek-r1","label":"DeepSeek R1","models":[{"id":"deepseek/deepseek-r1","name":"R1","releaseLabel":"R1","generationId":"r1","generationLabel":"R1","variantLabel":"Base","createdAt":"2025-01-20T13:51:35.000Z","kind":"language","contextLength":64000,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.7,"completionPerMillion":2.5,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"DeepSeek R1 is here: Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active in an inference pass....","huggingFaceId":"deepseek-ai/DeepSeek-R1","routes":[{"provider":"OpenRouter","modelId":"deepseek/deepseek-r1","url":"https://openrouter.ai/deepseek/deepseek-r1","free":false,"batch":false}],"benchmarks":[{"benchmark":"GPQA","score":71.5,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"MMLU-Pro","score":84,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"LEXam · open question","score":55.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://lexam-benchmark.github.io/","measuredAt":"2026-06-02","scope":"Registered model evaluation · LEXam Leaderboard"},{"benchmark":"LEXam · mcq 4 choices","score":52.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://lexam-benchmark.github.io/","measuredAt":"2026-06-02","scope":"Registered model evaluation · LEXam Leaderboard"},{"benchmark":"MMLU-Pro · mmlu pro","score":84,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-R1","measuredAt":"2026-01-28","scope":"Registered model evaluation · Model Card"},{"benchmark":"gpqa · diamond","score":71.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-R1","measuredAt":"2026-01-27","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"069c108b-a77f-4cce-bf04-28d58c2f91c1","title":"The State Of LLMs 2025","href":"/intel/069c108b-a77f-4cce-bf04-28d58c2f91c1","hook":"It distills a chaotic year of LLM research — DeepSeek's cost disruption, the rise of RLVR/GRPO for reasoning, and inference-time scaling — into one technical…"},{"id":"4425150c-cacb-4f7f-9b74-ec7df018ff9d","title":"A Technical Tour of the DeepSeek Models from V3 to V3.2","href":"/intel/4425150c-cacb-4f7f-9b74-ec7df018ff9d","hook":"If you're evaluating open-weight LLMs for agentic work, this walks through DeepSeek's architectural evolution from V3 to V3.2 — including the DeepSeek Sparse…"},{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"}]},{"id":"deepseek/deepseek-r1-distill-llama-70b","name":"R1 Distill Llama 70B","releaseLabel":"R1 Distill Llama 70B","generationId":"r1","generationLabel":"R1","variantLabel":"Distill Llama 70B","createdAt":"2025-01-23T20:12:49.000Z","kind":"language","contextLength":8192,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.7999999999999999,"completionPerMillion":0.7999999999999999,"imageOutput":null,"supportsTools":false,"supportsReasoning":true,"description":"DeepSeek R1 Distill Llama 70B is a distilled large language model based on [Llama-3.3-70B-Instruct](/meta-llama/llama-3.3-70b-instruct), using outputs from [DeepSeek R1](/deepseek/deepseek-r1). The model combines advanced distillation techniques to achieve high performance across...","huggingFaceId":"deepseek-ai/DeepSeek-R1-Distill-Llama-70B","routes":[{"provider":"OpenRouter","modelId":"deepseek/deepseek-r1-distill-llama-70b","url":"https://openrouter.ai/deepseek/deepseek-r1-distill-llama-70b","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"069c108b-a77f-4cce-bf04-28d58c2f91c1","title":"The State Of LLMs 2025","href":"/intel/069c108b-a77f-4cce-bf04-28d58c2f91c1","hook":"It distills a chaotic year of LLM research — DeepSeek's cost disruption, the rise of RLVR/GRPO for reasoning, and inference-time scaling — into one technical…"},{"id":"4425150c-cacb-4f7f-9b74-ec7df018ff9d","title":"A Technical Tour of the DeepSeek Models from V3 to V3.2","href":"/intel/4425150c-cacb-4f7f-9b74-ec7df018ff9d","hook":"If you're evaluating open-weight LLMs for agentic work, this walks through DeepSeek's architectural evolution from V3 to V3.2 — including the DeepSeek Sparse…"},{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"}]},{"id":"deepseek/deepseek-r1-0528","name":"R1 0528","releaseLabel":"R1 0528","generationId":"r1","generationLabel":"R1","variantLabel":"0528","createdAt":"2025-05-28T17:59:30.000Z","kind":"language","contextLength":163840,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.5,"completionPerMillion":2.1500000000000004,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...","huggingFaceId":"deepseek-ai/DeepSeek-R1-0528","routes":[{"provider":"OpenRouter","modelId":"deepseek/deepseek-r1-0528","url":"https://openrouter.ai/deepseek/deepseek-r1-0528","free":false,"batch":false}],"benchmarks":[{"benchmark":"MMLU-Pro","score":85,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"gpqa · diamond","score":78.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/a1/dd/a1dd2ea2-f092-423e-bc53-04c2fba5948d.json","measuredAt":"2026-04-17","scope":"Registered model evaluation · EvalEval"},{"benchmark":"MMLU-Pro · mmlu pro","score":85,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-R1-0528","measuredAt":"2026-01-28","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"069c108b-a77f-4cce-bf04-28d58c2f91c1","title":"The State Of LLMs 2025","href":"/intel/069c108b-a77f-4cce-bf04-28d58c2f91c1","hook":"It distills a chaotic year of LLM research — DeepSeek's cost disruption, the rise of RLVR/GRPO for reasoning, and inference-time scaling — into one technical…"},{"id":"4425150c-cacb-4f7f-9b74-ec7df018ff9d","title":"A Technical Tour of the DeepSeek Models from V3 to V3.2","href":"/intel/4425150c-cacb-4f7f-9b74-ec7df018ff9d","hook":"If you're evaluating open-weight LLMs for agentic work, this walks through DeepSeek's architectural evolution from V3 to V3.2 — including the DeepSeek Sparse…"},{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"}]}]}]},{"id":"qwen","name":"Qwen","domain":"qwen.ai","major":true,"modelCount":61,"series":[{"id":"qwen","label":"Qwen","models":[{"id":"qwen/qwen-2.5-72b-instruct","name":"Qwen2.5 72B Instruct","releaseLabel":"2.5 72B Instruct","generationId":"2.5","generationLabel":"2.5","variantLabel":"72B Instruct","createdAt":"2024-09-19T00:00:00.000Z","kind":"language","contextLength":32768,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.36,"completionPerMillion":0.39999999999999997,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Qwen2.5 72B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","huggingFaceId":"Qwen/Qwen2.5-72B-Instruct","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen-2.5-72b-instruct","url":"https://openrouter.ai/qwen/qwen-2.5-72b-instruct","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":53.9,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2024_08_31.csv","measuredAt":"2024-08-31","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1269,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 260 · 39,406 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen-2.5-7b-instruct","name":"Qwen2.5 7B Instruct","releaseLabel":"2.5 7B Instruct","generationId":"2.5","generationLabel":"2.5","variantLabel":"7B Instruct","createdAt":"2024-10-16T00:00:00.000Z","kind":"language","contextLength":32768,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.09999999999999999,"completionPerMillion":0.19999999999999998,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Qwen2.5 7B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","huggingFaceId":"Qwen/Qwen2.5-7B-Instruct","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen-2.5-7b-instruct","url":"https://openrouter.ai/qwen/qwen-2.5-7b-instruct","free":false,"batch":false}],"benchmarks":[{"benchmark":"signalbench · src","score":0.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation · signalbench raw per-item responses"},{"benchmark":"signalbench · time","score":0.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation"},{"benchmark":"signalbench · access deny","score":0.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation"},{"benchmark":"signalbench · memory label","score":0.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation"},{"benchmark":"signalbench · injection","score":0.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation"},{"benchmark":"signalbench · bot policy","score":0.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation"},{"benchmark":"LEXam · open question","score":16.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://lexam-benchmark.github.io/","measuredAt":"2026-06-02","scope":"Registered model evaluation · LEXam Leaderboard"},{"benchmark":"LEXam · mcq 4 choices","score":29.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://lexam-benchmark.github.io/","measuredAt":"2026-06-02","scope":"Registered model evaluation · LEXam Leaderboard"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen-plus-2025-01-25","name":"Qwen-Plus","releaseLabel":"Plus","generationId":"releases","generationLabel":"Releases","variantLabel":"Plus","createdAt":"2025-02-01T11:37:20.000Z","kind":"language","contextLength":1000000,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.26,"completionPerMillion":0.78,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Qwen-Plus, based on the Qwen2.5 foundation model, is a 131K context model with a balanced performance, speed, and cost combination.","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen-plus","url":"https://openrouter.ai/qwen/qwen-plus","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3-235b-a22b-04-28","name":"Qwen3 235B A22B","releaseLabel":"3 235B A22B","generationId":"3","generationLabel":"3","variantLabel":"235B A22B","createdAt":"2025-04-28T21:29:17.000Z","kind":"language","contextLength":131072,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.45499999999999996,"completionPerMillion":1.8199999999999998,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Qwen3-235B-A22B is a 235B parameter mixture-of-experts (MoE) model developed by Qwen, activating 22B parameters per forward pass. It supports seamless switching between a \"thinking\" mode for complex reasoning, math, and...","huggingFaceId":"Qwen/Qwen3-235B-A22B","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3-235b-a22b","url":"https://openrouter.ai/qwen/qwen3-235b-a22b","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":52.9,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1414,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 118 · 9,014 votes"},{"benchmark":"SWE-bench Pro","score":21.4,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"MMLU-Pro · mmlu pro","score":68.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/99/98/99989790-84de-4fd9-af37-1d9c28dd95b3.json","measuredAt":"2026-06-30","scope":"Registered model evaluation · EvalEval"},{"benchmark":"SWE-bench_Pro · SWE Bench Pro","score":21.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://scale.com/leaderboard/swe_bench_pro_public","measuredAt":"2026-02-28","scope":"Registered model evaluation · SWE-Bench Pro official evaluation results"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3-32b-04-28","name":"Qwen3 32B","releaseLabel":"3 32B","generationId":"3","generationLabel":"3","variantLabel":"32B","createdAt":"2025-04-28T21:32:25.000Z","kind":"language","contextLength":131072,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.08,"completionPerMillion":0.28,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","huggingFaceId":"Qwen/Qwen3-32B","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3-32b","url":"https://openrouter.ai/qwen/qwen3-32b","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":42.7,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1340,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 194 · 3,926 votes"},{"benchmark":"LEXam · open question","score":40,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://lexam-benchmark.github.io/","measuredAt":"2026-06-02","scope":"Registered model evaluation · LEXam Leaderboard"},{"benchmark":"LEXam · mcq 4 choices","score":45.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://lexam-benchmark.github.io/","measuredAt":"2026-06-02","scope":"Registered model evaluation · LEXam Leaderboard"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3-14b-04-28","name":"Qwen3 14B","releaseLabel":"3 14B","generationId":"3","generationLabel":"3","variantLabel":"14B","createdAt":"2025-04-28T21:41:18.000Z","kind":"language","contextLength":131072,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.12,"completionPerMillion":0.24,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Qwen3-14B is a dense 14.8B parameter causal language model from the Qwen3 series, designed for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","huggingFaceId":"Qwen/Qwen3-14B","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3-14b","url":"https://openrouter.ai/qwen/qwen3-14b","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":69.6,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2025_04_25.csv","measuredAt":"2025-04-25","scope":"Mean across that public release's task columns"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3-8b-04-28","name":"Qwen3 8B","releaseLabel":"3 8B","generationId":"3","generationLabel":"3","variantLabel":"8B","createdAt":"2025-04-28T21:43:52.000Z","kind":"language","contextLength":131072,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.117,"completionPerMillion":0.45499999999999996,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Qwen3-8B is a dense 8.2B parameter causal language model from the Qwen3 series, designed for both reasoning-heavy tasks and efficient dialogue. It supports seamless switching between \"thinking\" mode for math,...","huggingFaceId":"Qwen/Qwen3-8B","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3-8b","url":"https://openrouter.ai/qwen/qwen3-8b","free":false,"batch":false}],"benchmarks":[{"benchmark":"ifstruct-v1.0 · ifstruct v1","score":79.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://www.liquid.ai/blog/ifstruct-v1.0","measuredAt":"2026-06-30","scope":"Registered model evaluation · Liquid AI — IFStruct v1.0 blog (Qwen3-8B)"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3-30b-a3b-04-28","name":"Qwen3 30B A3B","releaseLabel":"3 30B A3B","generationId":"3","generationLabel":"3","variantLabel":"30B A3B","createdAt":"2025-04-28T22:16:44.000Z","kind":"language","contextLength":131072,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.12,"completionPerMillion":0.5,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique...","huggingFaceId":"Qwen/Qwen3-30B-A3B","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3-30b-a3b","url":"https://openrouter.ai/qwen/qwen3-30b-a3b","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":38.8,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1317,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 218 · 26,549 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3-235b-a22b-07-25","name":"Qwen3 235B A22B Instruct 2507","releaseLabel":"3 235B A22B Instruct 2507","generationId":"3","generationLabel":"3","variantLabel":"235B A22B 2507","createdAt":"2025-07-21T17:39:15.000Z","kind":"language","contextLength":262144,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.0875,"completionPerMillion":0.35,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...","huggingFaceId":"Qwen/Qwen3-235B-A22B-Instruct-2507","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3-235b-a22b-2507","url":"https://openrouter.ai/qwen/qwen3-235b-a22b-2507","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":48,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1419,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 104 · 97,398 votes"},{"benchmark":"MMLU-Pro · mmlu pro","score":83,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/df/98/df9861f7-59cf-4b5e-8988-571481f47d14.json","measuredAt":"2026-06-30","scope":"Registered model evaluation · EvalEval"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3-235b-a22b-thinking-2507","name":"Qwen3 235B A22B Thinking 2507","releaseLabel":"3 235B A22B Thinking 2507","generationId":"3","generationLabel":"3","variantLabel":"235B A22B Thinking 2507","createdAt":"2025-07-25T13:19:17.000Z","kind":"language","contextLength":131072,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.22999999999999998,"completionPerMillion":2.3,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144...","huggingFaceId":"Qwen/Qwen3-235B-A22B-Thinking-2507","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3-235b-a22b-thinking-2507","url":"https://openrouter.ai/qwen/qwen3-235b-a22b-thinking-2507","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":52.9,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1414,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 118 · 9,014 votes"},{"benchmark":"MMLU-Pro","score":84.4,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"LEXam · open question","score":47.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://lexam-benchmark.github.io/","measuredAt":"2026-06-02","scope":"Registered model evaluation · LEXam Leaderboard"},{"benchmark":"LEXam · mcq 4 choices","score":48.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://lexam-benchmark.github.io/","measuredAt":"2026-06-02","scope":"Registered model evaluation · LEXam Leaderboard"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3-30b-a3b-instruct-2507","name":"Qwen3 30B A3B Instruct 2507","releaseLabel":"3 30B A3B Instruct 2507","generationId":"3","generationLabel":"3","variantLabel":"30B A3B Instruct 2507","createdAt":"2025-07-29T16:36:05.000Z","kind":"language","contextLength":262144,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.04815,"completionPerMillion":0.19305,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...","huggingFaceId":"Qwen/Qwen3-30B-A3B-Instruct-2507","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3-30b-a3b-instruct-2507","url":"https://openrouter.ai/qwen/qwen3-30b-a3b-instruct-2507","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1384,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 150 · 23,828 votes"},{"benchmark":"gpqa · main","score":70.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3-30B-A3B-Instruct-2507","measuredAt":"2026-01-28","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3-30b-a3b-thinking-2507","name":"Qwen3 30B A3B Thinking 2507","releaseLabel":"3 30B A3B Thinking 2507","generationId":"3","generationLabel":"3","variantLabel":"30B A3B Thinking 2507","createdAt":"2025-08-28T16:39:52.000Z","kind":"language","contextLength":81920,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.19999999999999998,"completionPerMillion":2.4,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Qwen3-30B-A3B-Thinking-2507 is a 30B parameter Mixture-of-Experts reasoning model optimized for complex tasks requiring extended multi-step thinking. The model is designed specifically for “thinking mode,” where internal reasoning traces are separated...","huggingFaceId":"Qwen/Qwen3-30B-A3B-Thinking-2507","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3-30b-a3b-thinking-2507","url":"https://openrouter.ai/qwen/qwen3-30b-a3b-thinking-2507","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":38.8,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1317,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 218 · 26,549 votes"},{"benchmark":"AIME 2026","score":87.5,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"HMMT 2026","score":78.8,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"aime_2026 · MathArena/aime 2026","score":87.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://matharena.ai/?comp=aime--aime_2026","measuredAt":"2026-03-17","scope":"Registered model evaluation · Official MathArena Evaluation"},{"benchmark":"gpqa · diamond","score":72.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/4b/02/4b02c12c-3f7b-45c8-b746-c590860a7a36.json","measuredAt":"2026-04-17","scope":"Registered model evaluation · EvalEval"},{"benchmark":"MMLU-Pro · mmlu pro","score":80.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/46/a3/46a34cab-9475-40a1-ba07-2c0528f6bf5d.json","measuredAt":"2026-06-30","scope":"Registered model evaluation · EvalEval"},{"benchmark":"hmmt_feb_2026 · MathArena/hmmt feb 2026","score":78.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://matharena.ai/?comp=hmmt--hmmt_feb_2026","measuredAt":"2026-03-17","scope":"Registered model evaluation · Official MathArena Evaluation"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen-plus-2025-07-28","name":"Qwen Plus 0728","releaseLabel":"Plus 0728","generationId":"releases","generationLabel":"Releases","variantLabel":"Plus 0728","createdAt":"2025-09-08T16:06:39.000Z","kind":"language","contextLength":1000000,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.26,"completionPerMillion":0.78,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Qwen Plus 0728, based on the Qwen3 foundation model, is a 1 million context hybrid reasoning model with a balanced performance, speed, and cost combination.","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen-plus-2025-07-28","url":"https://openrouter.ai/qwen/qwen-plus-2025-07-28","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3-next-80b-a3b-instruct-2509","name":"Qwen3 Next 80B A3B Instruct","releaseLabel":"3 Next 80B A3B Instruct","generationId":"3","generationLabel":"3","variantLabel":"Next 80B A3B Instruct","createdAt":"2025-09-11T17:36:53.000Z","kind":"language","contextLength":262144,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.09999999999999999,"completionPerMillion":1.1,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","huggingFaceId":"Qwen/Qwen3-Next-80B-A3B-Instruct","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3-next-80b-a3b-instruct","url":"https://openrouter.ai/qwen/qwen3-next-80b-a3b-instruct","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":47.4,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1419,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 106 · 23,000 votes"},{"benchmark":"MMLU-Pro","score":80.6,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"signalbench · src","score":0.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation · signalbench raw per-item responses"},{"benchmark":"signalbench · time","score":0.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation"},{"benchmark":"signalbench · access deny","score":0.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation"},{"benchmark":"signalbench · memory label","score":0.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation"},{"benchmark":"signalbench · injection","score":0.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation"},{"benchmark":"signalbench · bot policy","score":0.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation"},{"benchmark":"gpqa · gpqa diamond","score":72.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3-Next-80B-A3B-Instruct","measuredAt":"2026-01-15","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3-next-80b-a3b-thinking-2509","name":"Qwen3 Next 80B A3B Thinking","releaseLabel":"3 Next 80B A3B Thinking","generationId":"3","generationLabel":"3","variantLabel":"Next 80B A3B Thinking","createdAt":"2025-09-11T17:38:04.000Z","kind":"language","contextLength":262144,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.15,"completionPerMillion":1.2,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Qwen3-Next-80B-A3B-Thinking is a reasoning-first chat model in the Qwen3-Next line that outputs structured “thinking” traces by default. It’s designed for hard multi-step problems; math proofs, code synthesis/debugging, logic, and agentic...","huggingFaceId":"Qwen/Qwen3-Next-80B-A3B-Thinking","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3-next-80b-a3b-thinking","url":"https://openrouter.ai/qwen/qwen3-next-80b-a3b-thinking","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":51,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1368,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 167 · 13,749 votes"},{"benchmark":"LEXam · open question","score":43.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://lexam-benchmark.github.io/","measuredAt":"2026-06-02","scope":"Registered model evaluation · LEXam Leaderboard"},{"benchmark":"LEXam · mcq 4 choices","score":43.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://lexam-benchmark.github.io/","measuredAt":"2026-06-02","scope":"Registered model evaluation · LEXam Leaderboard"},{"benchmark":"gpqa · gpqa diamond","score":73.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3-Next-80B-A3B-Thinking","measuredAt":"2026-01-16","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3-max","name":"Qwen3 Max","releaseLabel":"3 Max","generationId":"3","generationLabel":"3","variantLabel":"Max","createdAt":"2025-09-23T21:26:48.000Z","kind":"language","contextLength":262144,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.78,"completionPerMillion":3.9,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Qwen3-Max is an updated release built on the Qwen3 series, offering major improvements in reasoning, instruction following, multilingual support, and long-tail knowledge coverage compared to the January 2025 version. It...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3-max","url":"https://openrouter.ai/qwen/qwen3-max","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1439,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 65 · 27,838 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3-max-thinking-20260123","name":"Qwen3 Max Thinking","releaseLabel":"3 Max Thinking","generationId":"3","generationLabel":"3","variantLabel":"Max Thinking","createdAt":"2026-02-09T21:18:21.000Z","kind":"language","contextLength":262144,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.78,"completionPerMillion":3.9,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Qwen3-Max-Thinking is the flagship reasoning model in the Qwen3 series, designed for high-stakes cognitive tasks that require deep, multi-step reasoning. By significantly scaling model capacity and reinforcement learning compute, it...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3-max-thinking","url":"https://openrouter.ai/qwen/qwen3-max-thinking","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1439,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 65 · 27,838 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3.5-397b-a17b-20260216","name":"Qwen3.5 397B A17B","releaseLabel":"3.5 397B A17B","generationId":"3.5","generationLabel":"3.5","variantLabel":"397B A17B","createdAt":"2026-02-16T06:23:38.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["text","image","video"],"outputModalities":["text"],"promptPerMillion":0.39,"completionPerMillion":2.34,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...","huggingFaceId":"Qwen/Qwen3.5-397B-A17B","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3.5-397b-a17b","url":"https://openrouter.ai/qwen/qwen3.5-397b-a17b","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1438,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 67 · 70,207 votes"},{"benchmark":"AIME 2026","score":93.3,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"GPQA","score":88.4,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"Humanity's Last Exam","score":28.7,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"HMMT 2026","score":87.9,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"MMLU-Pro","score":87.8,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"SWE-bench Verified","score":76.4,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"Terminal-Bench","score":52.5,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"SWE-bench_Multilingual · swe bench multilingual % resolved","score":69.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.5-397B-A17B","measuredAt":"2026-08-10","scope":"Registered model evaluation · Model Card"},{"benchmark":"WildClawBench · overall","score":34.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://internlm.github.io/WildClawBench","measuredAt":"2026-05-22","scope":"Registered model evaluation · WildClawBench"},{"benchmark":"ResearchClawBench · overall","score":14.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.5-397B-A17B","measuredAt":"2026-04-16","scope":"Registered model evaluation · Model Card"},{"benchmark":"Claw-Eval · general","score":57.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://claw-eval.github.io","measuredAt":"2026-04-23","scope":"Registered model evaluation · Claw-Eval Leaderboard"},{"benchmark":"Claw-Eval · multimodal","score":20.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://claw-eval.github.io","measuredAt":"2026-04-23","scope":"Registered model evaluation · Claw-Eval Leaderboard"},{"benchmark":"Claw-Eval · multi turn","score":52.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://claw-eval.github.io","measuredAt":"2026-04-23","scope":"Registered model evaluation · Claw-Eval Leaderboard"},{"benchmark":"Video-MME-v2 · video-mme-v2","score":55.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://video-mme-v2.netlify.app/#leaderboard","measuredAt":"2026-04-30","scope":"Registered model evaluation · VideoMMEv2 leaderboard"},{"benchmark":"apex-agents · apex-agents","score":13.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://www.mercor.com/apex/apex-agents-leaderboard/","measuredAt":"2026-04-22","scope":"Registered model evaluation · APEX-Agents Leaderboard"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3.5-plus-20260216","name":"Qwen3.5 Plus 2026-02-15","releaseLabel":"3.5 Plus 2026-02-15","generationId":"3.5","generationLabel":"3.5","variantLabel":"Plus · 02-15","createdAt":"2026-02-16T08:10:16.000Z","kind":"multimodal","contextLength":1000000,"inputModalities":["text","image","video"],"outputModalities":["text"],"promptPerMillion":0.26,"completionPerMillion":1.56,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"The Qwen3.5 native vision-language series Plus models are built on a hybrid architecture that integrates linear attention mechanisms with sparse mixture-of-experts models, achieving higher inference efficiency. In a variety of...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3.5-plus-02-15","url":"https://openrouter.ai/qwen/qwen3.5-plus-02-15","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3.5-flash-20260224","name":"Qwen3.5-Flash","releaseLabel":"3.5-Flash","generationId":"3.5","generationLabel":"3.5","variantLabel":"Flash · 02-23","createdAt":"2026-02-25T21:09:36.000Z","kind":"multimodal","contextLength":1000000,"inputModalities":["text","image","video"],"outputModalities":["text"],"promptPerMillion":0.065,"completionPerMillion":0.26,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"The Qwen3.5 native vision-language Flash models are built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. Compared to the...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3.5-flash-02-23","url":"https://openrouter.ai/qwen/qwen3.5-flash-02-23","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1398,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 140 · 58,319 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3.5-122b-a10b-20260224","name":"Qwen3.5-122B-A10B","releaseLabel":"3.5-122B-A10B","generationId":"3.5","generationLabel":"3.5","variantLabel":"122B A10B","createdAt":"2026-02-25T21:09:49.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["text","image","video"],"outputModalities":["text"],"promptPerMillion":0.26,"completionPerMillion":2.08,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...","huggingFaceId":"Qwen/Qwen3.5-122B-A10B","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3.5-122b-a10b","url":"https://openrouter.ai/qwen/qwen3.5-122b-a10b","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1418,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 108 · 28,413 votes"},{"benchmark":"GPQA","score":86.6,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"Humanity's Last Exam","score":25.3,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"MMLU-Pro","score":86.7,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"SWE-bench Verified","score":72,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"Terminal-Bench","score":49.4,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"gpqa · diamond","score":84.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/58/60/5860f5bd-926e-4ece-bc87-a1af551a6280.json","measuredAt":"2026-04-20","scope":"Registered model evaluation · EvalEval"},{"benchmark":"MMLU-Pro · mmlu pro","score":86.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/af/3c/af3c554a-1eb9-4a79-80dd-678c7aebf0aa.json","measuredAt":"2026-06-30","scope":"Registered model evaluation · EvalEval"},{"benchmark":"yc-bench · medium","score":0,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/collinear-ai/yc-bench","measuredAt":"2026-04-02","scope":"Registered model evaluation · YC-Bench eval"},{"benchmark":"ScreenSpot-Pro · overall","score":70.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.5-122B-A10B","measuredAt":"2026-03-18","scope":"Registered model evaluation · Model Card"},{"benchmark":"hle · hle","score":25.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.5-122B-A10B","measuredAt":"2026-02-25","scope":"Registered model evaluation · Model Card"},{"benchmark":"SWE-bench_Verified · swe bench % resolved","score":72,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.5-122B-A10B","measuredAt":"2026-02-03","scope":"Registered model evaluation · Model Card"},{"benchmark":"terminal-bench-2.0 · terminalbench 2","score":49.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.5-122B-A10B","measuredAt":"2026-02-25","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3.5-27b-20260224","name":"Qwen3.5-27B","releaseLabel":"3.5-27B","generationId":"3.5","generationLabel":"3.5","variantLabel":"27B","createdAt":"2026-02-25T21:10:10.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["text","image","video"],"outputModalities":["text"],"promptPerMillion":0.195,"completionPerMillion":1.56,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...","huggingFaceId":"Qwen/Qwen3.5-27B","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3.5-27b","url":"https://openrouter.ai/qwen/qwen3.5-27b","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1408,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 128 · 27,282 votes"},{"benchmark":"AIME 2026","score":90.8,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"GPQA","score":85.5,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"Humanity's Last Exam","score":24.3,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"HMMT 2026","score":81.1,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"MMLU-Pro","score":86.1,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"SWE-bench Verified","score":72.4,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"Terminal-Bench","score":41.6,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"gpqa · diamond","score":81.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/5b/a4/5ba4cf19-18b3-4d08-bad0-ef29422cb18a.json","measuredAt":"2026-04-20","scope":"Registered model evaluation · EvalEval"},{"benchmark":"MMLU-Pro · mmlu pro","score":86.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/15/df/15df0778-3ca6-4ff9-9b34-41f74dc4ced4.json","measuredAt":"2026-06-30","scope":"Registered model evaluation · EvalEval"},{"benchmark":"MMMU_Pro · mmmu pro vision","score":75,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.5-27B","measuredAt":"2026-04-28","scope":"Registered model evaluation · Model Card"},{"benchmark":"ScreenSpot-Pro · overall","score":70.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.5-27B","measuredAt":"2026-03-18","scope":"Registered model evaluation · Model Card"},{"benchmark":"hmmt_feb_2026 · MathArena/hmmt feb 2026","score":81.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://matharena.ai/?comp=hmmt--hmmt_feb_2026","measuredAt":"2026-03-17","scope":"Registered model evaluation · Official MathArena Evaluation"},{"benchmark":"aime_2026 · MathArena/aime 2026","score":90.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://matharena.ai/?comp=aime--aime_2026","measuredAt":"2026-03-17","scope":"Registered model evaluation · Official MathArena Evaluation"},{"benchmark":"hle · hle","score":24.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.5-27B","measuredAt":"2026-02-25","scope":"Registered model evaluation · Model Card"},{"benchmark":"SWE-bench_Verified · swe bench % resolved","score":72.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.5-27B","measuredAt":"2026-02-03","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3.5-35b-a3b-20260224","name":"Qwen3.5-35B-A3B","releaseLabel":"3.5-35B-A3B","generationId":"3.5","generationLabel":"3.5","variantLabel":"35B A3B","createdAt":"2026-02-25T21:10:22.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["text","image","video"],"outputModalities":["text"],"promptPerMillion":0.25,"completionPerMillion":1.25,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...","huggingFaceId":"Qwen/Qwen3.5-35B-A3B","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3.5-35b-a3b","url":"https://openrouter.ai/qwen/qwen3.5-35b-a3b","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1396,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 142 · 29,097 votes"},{"benchmark":"AIME 2026","score":93.3,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"GPQA","score":84.2,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"Humanity's Last Exam","score":22.4,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"HMMT 2026","score":81.8,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"MMLU-Pro","score":85.3,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"SWE-bench Verified","score":69.2,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"Terminal-Bench","score":40.5,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"ExtractBench · mean","score":88,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/llamaindex/ExtractBench","measuredAt":"2026-08-24","scope":"Registered model evaluation · ExtractBench"},{"benchmark":"ExtractBench · short","score":93.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/llamaindex/ExtractBench","measuredAt":"2026-08-24","scope":"Registered model evaluation · ExtractBench"},{"benchmark":"ExtractBench · medium","score":85.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/llamaindex/ExtractBench","measuredAt":"2026-08-24","scope":"Registered model evaluation · ExtractBench"},{"benchmark":"ExtractBench · long","score":31.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/llamaindex/ExtractBench","measuredAt":"2026-08-24","scope":"Registered model evaluation · ExtractBench"},{"benchmark":"gpqa · diamond","score":85.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/e6/75/e675e8b1-162d-4f07-9f01-baef2cb8e59c.json","measuredAt":"2026-04-20","scope":"Registered model evaluation · EvalEval"},{"benchmark":"MMLU-Pro · mmlu pro","score":85.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/3e/3c/3e3cd8f4-e9ba-448f-b4fe-bf755ef9b311.json","measuredAt":"2026-06-30","scope":"Registered model evaluation · EvalEval"},{"benchmark":"MMMU_Pro · mmmu pro vision","score":75.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.5-35B-A3B","measuredAt":"2026-04-28","scope":"Registered model evaluation · Model Card"},{"benchmark":"ScreenSpot-Pro · overall","score":68.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.5-35B-A3B","measuredAt":"2026-03-18","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3.5-9b-20260310","name":"Qwen3.5-9B","releaseLabel":"3.5-9B","generationId":"3.5","generationLabel":"3.5","variantLabel":"9B","createdAt":"2026-03-10T14:19:56.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["text","image","video"],"outputModalities":["text"],"promptPerMillion":0.09999999999999999,"completionPerMillion":0.15,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...","huggingFaceId":"Qwen/Qwen3.5-9B","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3.5-9b","url":"https://openrouter.ai/qwen/qwen3.5-9b","free":false,"batch":false}],"benchmarks":[{"benchmark":"AIME 2026","score":92.5,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"GPQA","score":81.7,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"HMMT 2026","score":71.2,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"MMLU-Pro","score":82.5,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"MDPBench · overall","score":65.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/Delores-Lin/MDPBench","measuredAt":"2026-04-14","scope":"Registered model evaluation · MDPBench leaderboard"},{"benchmark":"MDPBench · digital","score":74.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/Delores-Lin/MDPBench","measuredAt":"2026-04-14","scope":"Registered model evaluation · MDPBench leaderboard"},{"benchmark":"MDPBench · photographed","score":62.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/Delores-Lin/MDPBench","measuredAt":"2026-04-14","scope":"Registered model evaluation · MDPBench leaderboard"},{"benchmark":"MDPBench · latin","score":72.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/Delores-Lin/MDPBench","measuredAt":"2026-04-14","scope":"Registered model evaluation · MDPBench leaderboard"},{"benchmark":"MDPBench · de","score":72.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/Delores-Lin/MDPBench","measuredAt":"2026-04-14","scope":"Registered model evaluation · MDPBench leaderboard"},{"benchmark":"MDPBench · en","score":72,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/Delores-Lin/MDPBench","measuredAt":"2026-04-14","scope":"Registered model evaluation · MDPBench leaderboard"},{"benchmark":"MDPBench · es","score":72,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/Delores-Lin/MDPBench","measuredAt":"2026-04-14","scope":"Registered model evaluation · MDPBench leaderboard"},{"benchmark":"MDPBench · fr","score":64.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/Delores-Lin/MDPBench","measuredAt":"2026-04-14","scope":"Registered model evaluation · MDPBench leaderboard"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3.6-plus-04-02","name":"Qwen3.6 Plus","releaseLabel":"3.6 Plus","generationId":"3.6","generationLabel":"3.6","variantLabel":"Plus","createdAt":"2026-04-02T12:39:17.000Z","kind":"multimodal","contextLength":1000000,"inputModalities":["text","image","video"],"outputModalities":["text"],"promptPerMillion":0.325,"completionPerMillion":1.95,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Qwen 3.6 Plus builds on a hybrid architecture that combines efficient linear attention with sparse mixture-of-experts routing, enabling strong scalability and high-performance inference. Compared to the 3.5 series, it delivers...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3.6-plus","url":"https://openrouter.ai/qwen/qwen3.6-plus","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":69,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1437,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 70 · 45,376 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3.6-27b-20260422","name":"Qwen3.6 27B","releaseLabel":"3.6 27B","generationId":"3.6","generationLabel":"3.6","variantLabel":"27B","createdAt":"2026-04-27T01:57:44.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["text","image","video"],"outputModalities":["text"],"promptPerMillion":0.6,"completionPerMillion":3.5999999999999996,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...","huggingFaceId":"Qwen/Qwen3.6-27B","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3.6-27b","url":"https://openrouter.ai/qwen/qwen3.6-27b","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":64.2,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"skillsbench · skillsbench v1 1","score":48.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.6-27B","measuredAt":"2026-04-21","scope":"Registered model evaluation · Qwen3.6-27B model card — Benchmark Results (Language > Coding Agent)"},{"benchmark":"terminal-bench-2.0 · terminalbench 2","score":59.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.6-27B","measuredAt":"2026-06-23","scope":"Registered model evaluation · Model Card"},{"benchmark":"WildClawBench · overall","score":43.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://internlm.github.io/WildClawBench","measuredAt":"2026-08-11","scope":"Registered model evaluation · WildClawBench"},{"benchmark":"SWE-bench_Multilingual · swe bench multilingual % resolved","score":71.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.6-27B","measuredAt":"2026-08-10","scope":"Registered model evaluation · Model Card"},{"benchmark":"MMMU_Pro · mmmu pro vision","score":75.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.6-27B","measuredAt":"2026-05-15","scope":"Registered model evaluation · Model Card"},{"benchmark":"aime_2026 · MathArena/aime 2026","score":94.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.6-27B","measuredAt":"2026-04-22","scope":"Registered model evaluation · Model Card"},{"benchmark":"gpqa · diamond","score":87.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.6-27B","measuredAt":"2026-04-22","scope":"Registered model evaluation · Model Card"},{"benchmark":"hle · hle","score":24,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.6-27B","measuredAt":"2026-04-22","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3.6-max-preview-20260420","name":"Qwen3.6 Max Preview","releaseLabel":"3.6 Max Preview","generationId":"3.6","generationLabel":"3.6","variantLabel":"Max Preview","createdAt":"2026-04-27T03:24:02.000Z","kind":"language","contextLength":262144,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":1.0270000000000001,"completionPerMillion":6.162,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Qwen3.6-Max-Preview is a proprietary frontier model from Alibaba Cloud built on a sparse mixture-of-experts architecture with approximately 1 trillion total parameters. It is optimized for agentic coding, tool use, and...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3.6-max-preview","url":"https://openrouter.ai/qwen/qwen3.6-max-preview","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1446,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 46 · 5,196 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3.6-35b-a3b-20260415","name":"Qwen3.6 35B A3B","releaseLabel":"3.6 35B A3B","generationId":"3.6","generationLabel":"3.6","variantLabel":"35B A3B","createdAt":"2026-04-27T03:24:15.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["text","image","video"],"outputModalities":["text"],"promptPerMillion":0.09999999999999999,"completionPerMillion":0.8999999999999999,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...","huggingFaceId":"Qwen/Qwen3.6-35B-A3B","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3.6-35b-a3b","url":"https://openrouter.ai/qwen/qwen3.6-35b-a3b","free":false,"batch":false}],"benchmarks":[{"benchmark":"aime_2026 · MathArena/aime 2026","score":92.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.6-35B-A3B","measuredAt":"2026-04-16","scope":"Registered model evaluation · Model Card"},{"benchmark":"gpqa · diamond","score":86,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.6-35B-A3B","measuredAt":"2026-04-16","scope":"Registered model evaluation · Model Card"},{"benchmark":"hle · hle","score":21.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.6-35B-A3B","measuredAt":"2026-04-16","scope":"Registered model evaluation · Model Card"},{"benchmark":"hmmt_feb_2026 · MathArena/hmmt feb 2026","score":83.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.6-35B-A3B","measuredAt":"2026-04-16","scope":"Registered model evaluation · Model Card"},{"benchmark":"MMLU-Pro · mmlu pro","score":85.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.6-35B-A3B","measuredAt":"2026-04-16","scope":"Registered model evaluation · Model Card"},{"benchmark":"SWE-bench_Pro · SWE Bench Pro","score":49.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.6-35B-A3B","measuredAt":"2026-04-16","scope":"Registered model evaluation · Model Card"},{"benchmark":"SWE-bench_Verified · swe bench % resolved","score":73.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.6-35B-A3B","measuredAt":"2026-04-16","scope":"Registered model evaluation · Model Card"},{"benchmark":"terminal-bench-2.0 · terminalbench 2","score":51.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.6-35B-A3B","measuredAt":"2026-04-16","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3.6-flash","name":"Qwen3.6 Flash","releaseLabel":"3.6 Flash","generationId":"3.6","generationLabel":"3.6","variantLabel":"Flash","createdAt":"2026-04-27T03:42:42.000Z","kind":"multimodal","contextLength":1000000,"inputModalities":["text","image","video"],"outputModalities":["text"],"promptPerMillion":0.1875,"completionPerMillion":1.125,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Qwen3.6 Flash is a fast, efficient language model from Alibaba's Qwen 3.6 series. It supports text, image, and video input with a 1M token context window. Tiered pricing kicks in...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3.6-flash","url":"https://openrouter.ai/qwen/qwen3.6-flash","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":60.5,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3.5-plus-20260420","name":"Qwen3.5 Plus 2026-04-20","releaseLabel":"3.5 Plus 2026-04-20","generationId":"3.5","generationLabel":"3.5","variantLabel":"Plus · 20260420","createdAt":"2026-04-27T03:42:48.000Z","kind":"multimodal","contextLength":1000000,"inputModalities":["text","image","video"],"outputModalities":["text"],"promptPerMillion":0.3,"completionPerMillion":1.7999999999999998,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Qwen3.5 Plus (April 2026) is a large-scale multimodal language model from Alibaba. It accepts text, image, and video input and produces text output, with a 1M token context window. This...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3.5-plus-20260420","url":"https://openrouter.ai/qwen/qwen3.5-plus-20260420","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3.7-max-20260520","name":"Qwen3.7 Max","releaseLabel":"3.7 Max","generationId":"3.7","generationLabel":"3.7","variantLabel":"Max","createdAt":"2026-05-21T15:21:01.000Z","kind":"language","contextLength":1000000,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":1.475,"completionPerMillion":4.425,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Qwen3.7-Max is the flagship model in Alibaba's Qwen3.7 series. It supports text input and output and is designed for agent-centric workloads, with particular strengths in coding, office and productivity tasks,...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3.7-max","url":"https://openrouter.ai/qwen/qwen3.7-max","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":74.1,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1474,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 19 · 3,712 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3.7-plus-20260602","name":"Qwen3.7 Plus","releaseLabel":"3.7 Plus","generationId":"3.7","generationLabel":"3.7","variantLabel":"Plus","createdAt":"2026-06-03T13:03:03.000Z","kind":"multimodal","contextLength":1000000,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0.32,"completionPerMillion":1.28,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Qwen3.7-Plus is a cost-effective model in Alibaba's Qwen3.7 series. It supports text and image input with text output, building on the series' text capabilities with a comprehensive upgrade to its...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3.7-plus","url":"https://openrouter.ai/qwen/qwen3.7-plus","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1456,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 33 · 34,426 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3.7-flash-20260727","name":"Qwen3.7 Flash","releaseLabel":"3.7 Flash","generationId":"3.7","generationLabel":"3.7","variantLabel":"Flash","createdAt":"2026-07-27T22:16:01.000Z","kind":"multimodal","contextLength":1000000,"inputModalities":["text","image","video"],"outputModalities":["text"],"promptPerMillion":0.03,"completionPerMillion":0.13,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Qwen3.7 Flash is a vision-language reasoning model from Alibaba. It is suited for multimodal agents, visual coding, search, and computer interaction, with strengths in object recognition, spatial understanding, and real-world...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3.7-flash","url":"https://openrouter.ai/qwen/qwen3.7-flash","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3.8-max-20260803","name":"Qwen3.8 Max","releaseLabel":"3.8 Max","generationId":"3.8","generationLabel":"3.8","variantLabel":"Max","createdAt":"2026-08-03T04:33:32.000Z","kind":"multimodal","contextLength":1000000,"inputModalities":["text","image","video"],"outputModalities":["text"],"promptPerMillion":2,"completionPerMillion":6,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Qwen3.8 Max is the flagship model in Alibaba's Qwen3.8 series, the general-availability successor to the Qwen3.8 Max Preview. It is a multimodal reasoning model intended for complex reasoning, visual understanding,...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3.8-max","url":"https://openrouter.ai/qwen/qwen3.8-max","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":79.5,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1482,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 11 · 9,955 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3.8-2.4t-a95b-20260812","name":"Qwen3.8 2.4T A95B","releaseLabel":"3.8 2.4T A95B","generationId":"3.8","generationLabel":"3.8","variantLabel":"2.4t A95B","createdAt":"2026-08-12T16:21:42.000Z","kind":"language","contextLength":1048576,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":2,"completionPerMillion":6,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of [Qwen3.8 Max](/qwen/qwen3.8-max), with 95 billion active parameters out of 2.4 trillion total. It is...","huggingFaceId":"Qwen/Qwen3.8-2.4T-A95B","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3.8-2.4t-a95b","url":"https://openrouter.ai/qwen/qwen3.8-2.4t-a95b","free":false,"batch":false}],"benchmarks":[{"benchmark":"AA Intelligence Index","score":57.7,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/qwen3-8-2-4t-a95b","measuredAt":"2026-08-27","scope":"Current public index; highest tested effort configuration"},{"benchmark":"deep-swe · deep swe","score":56.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","measuredAt":"2026-08-12","scope":"Registered model evaluation · Model Card"},{"benchmark":"gpqa · diamond","score":92.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","measuredAt":"2026-08-12","scope":"Registered model evaluation · Model Card"},{"benchmark":"hle · hle","score":43.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","measuredAt":"2026-08-12","scope":"Registered model evaluation · Model Card"},{"benchmark":"SWE-bench_Pro · SWE Bench Pro","score":67.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","measuredAt":"2026-08-12","scope":"Registered model evaluation · Model Card"},{"benchmark":"WildClawBench · overall","score":56.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://internlm.github.io/WildClawBench","measuredAt":"2026-08-12","scope":"Registered model evaluation · WildClawBench"},{"benchmark":"terminal-bench-2.1 · terminalbench 2 1","score":86.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","measuredAt":"2026-08-12","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3.8-27b-20260814","name":"Qwen3.8 27B","releaseLabel":"3.8 27B","generationId":"3.8","generationLabel":"3.8","variantLabel":"27B","createdAt":"2026-08-14T15:55:10.000Z","kind":"multimodal","contextLength":1000000,"inputModalities":["text","image","video"],"outputModalities":["text"],"promptPerMillion":0.425,"completionPerMillion":2.5500000000000003,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...","huggingFaceId":"Qwen/Qwen3.8-27B","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3.8-27b","url":"https://openrouter.ai/qwen/qwen3.8-27b","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":75.8,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"AA Intelligence Index","score":52,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/qwen3-8-27b","measuredAt":"2026-08-27","scope":"Current public index; highest tested effort configuration"},{"benchmark":"LMArena overall","score":1441,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 61 · 3,205 votes"},{"benchmark":"SWE-bench_Pro · SWE Bench Pro","score":61.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-27B","measuredAt":"2026-08-14","scope":"Registered model evaluation · Qwen3.8-27B model card"},{"benchmark":"deep-swe · deep swe","score":42.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-27B","measuredAt":"2026-08-14","scope":"Registered model evaluation · Qwen3.8-27B model card"},{"benchmark":"gpqa · diamond","score":89.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-27B","measuredAt":"2026-08-14","scope":"Registered model evaluation · Qwen3.8-27B model card"},{"benchmark":"hle · hle","score":30.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-27B","measuredAt":"2026-08-14","scope":"Registered model evaluation · Qwen3.8-27B model card"},{"benchmark":"Claw-Eval · multimodal","score":57.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-27B","measuredAt":"2026-08-14","scope":"Registered model evaluation · Qwen3.8-27B model card"},{"benchmark":"ExtractBench · mean","score":89.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/llamaindex/ExtractBench","measuredAt":"2026-08-24","scope":"Registered model evaluation · ExtractBench"},{"benchmark":"ExtractBench · short","score":94.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/llamaindex/ExtractBench","measuredAt":"2026-08-24","scope":"Registered model evaluation · ExtractBench"},{"benchmark":"ExtractBench · medium","score":87.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/llamaindex/ExtractBench","measuredAt":"2026-08-24","scope":"Registered model evaluation · ExtractBench"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3.8-flash-20260826","name":"Qwen3.8 Flash","releaseLabel":"3.8 Flash","generationId":"3.8","generationLabel":"3.8","variantLabel":"Flash","createdAt":"2026-08-26T19:37:40.000Z","kind":"multimodal","contextLength":1000000,"inputModalities":["text","image","video"],"outputModalities":["text"],"promptPerMillion":0.15,"completionPerMillion":0.47,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Qwen3.8 Flash is a multimodal reasoning model from Alibaba. It is suited for coding assistance, agentic workflows, visual understanding, document and codebase analysis, desktop interaction, chart analysis, and long-video analysis.","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3.8-flash","url":"https://openrouter.ai/qwen/qwen3.8-flash","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]}]},{"id":"qwen-vl","label":"Qwen VL","models":[{"id":"qwen/qwen2.5-vl-72b-instruct","name":"Qwen2.5 VL 72B Instruct","releaseLabel":"Qwen2.5 VL 72B Instruct","generationId":"2.5","generationLabel":"2.5","variantLabel":"72B Instruct","createdAt":"2025-02-01T11:45:11.000Z","kind":"multimodal","contextLength":128000,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0.25,"completionPerMillion":0.75,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Qwen2.5-VL is proficient in recognizing common objects such as flowers, birds, fish, and insects. It is also highly capable of analyzing texts, charts, icons, graphics, and layouts within images.","huggingFaceId":"Qwen/Qwen2.5-VL-72B-Instruct","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen2.5-vl-72b-instruct","url":"https://openrouter.ai/qwen/qwen2.5-vl-72b-instruct","free":false,"batch":false}],"benchmarks":[{"benchmark":"Real5-OmniDocBench · overall","score":86.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/PaddlePaddle/Real5-OmniDocBench","measuredAt":"2026-08-08","scope":"Registered model evaluation · Real5-OmniDocBench Leaderboard"},{"benchmark":"Real5-OmniDocBench · scanning","score":86.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/PaddlePaddle/Real5-OmniDocBench","measuredAt":"2026-08-08","scope":"Registered model evaluation · Real5-OmniDocBench Leaderboard"},{"benchmark":"Real5-OmniDocBench · warping","score":87.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/PaddlePaddle/Real5-OmniDocBench","measuredAt":"2026-08-08","scope":"Registered model evaluation · Real5-OmniDocBench Leaderboard"},{"benchmark":"Real5-OmniDocBench · screen photography","score":86.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/PaddlePaddle/Real5-OmniDocBench","measuredAt":"2026-08-08","scope":"Registered model evaluation · Real5-OmniDocBench Leaderboard"},{"benchmark":"Real5-OmniDocBench · illumination","score":87.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/PaddlePaddle/Real5-OmniDocBench","measuredAt":"2026-08-08","scope":"Registered model evaluation · Real5-OmniDocBench Leaderboard"},{"benchmark":"Real5-OmniDocBench · skew","score":86.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/PaddlePaddle/Real5-OmniDocBench","measuredAt":"2026-08-08","scope":"Registered model evaluation · Real5-OmniDocBench Leaderboard"},{"benchmark":"MMLU-Pro · mmlu pro","score":0.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/9c/45/9c455b8c-1153-4c24-9de0-1d506297d805.json","measuredAt":"2025-04-01","scope":"Registered model evaluation · EvalEval"},{"benchmark":"ScreenSpot-Pro · overall","score":53.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://gui-agent.github.io/grounding-leaderboard/","measuredAt":"2026-03-17","scope":"Registered model evaluation · ScreenSpot-Pro Leaderboard"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct","releaseLabel":"Qwen3 VL 235B A22B Instruct","generationId":"3","generationLabel":"3","variantLabel":"235B A22B Instruct","createdAt":"2025-09-23T23:04:47.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0.21,"completionPerMillion":1.9,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...","huggingFaceId":"Qwen/Qwen3-VL-235B-A22B-Instruct","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3-vl-235b-a22b-instruct","url":"https://openrouter.ai/qwen/qwen3-vl-235b-a22b-instruct","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1421,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 98 · 11,556 votes"},{"benchmark":"Real5-OmniDocBench · overall","score":88.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/PaddlePaddle/Real5-OmniDocBench","measuredAt":"2026-08-08","scope":"Registered model evaluation · Real5-OmniDocBench Leaderboard"},{"benchmark":"Real5-OmniDocBench · scanning","score":89.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/PaddlePaddle/Real5-OmniDocBench","measuredAt":"2026-08-08","scope":"Registered model evaluation · Real5-OmniDocBench Leaderboard"},{"benchmark":"Real5-OmniDocBench · warping","score":90,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/PaddlePaddle/Real5-OmniDocBench","measuredAt":"2026-08-08","scope":"Registered model evaluation · Real5-OmniDocBench Leaderboard"},{"benchmark":"Real5-OmniDocBench · screen photography","score":89.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/PaddlePaddle/Real5-OmniDocBench","measuredAt":"2026-08-08","scope":"Registered model evaluation · Real5-OmniDocBench Leaderboard"},{"benchmark":"Real5-OmniDocBench · illumination","score":89.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/PaddlePaddle/Real5-OmniDocBench","measuredAt":"2026-08-08","scope":"Registered model evaluation · Real5-OmniDocBench Leaderboard"},{"benchmark":"Real5-OmniDocBench · skew","score":86.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/PaddlePaddle/Real5-OmniDocBench","measuredAt":"2026-08-08","scope":"Registered model evaluation · Real5-OmniDocBench Leaderboard"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3-vl-235b-a22b-thinking","name":"Qwen3 VL 235B A22B Thinking","releaseLabel":"Qwen3 VL 235B A22B Thinking","generationId":"3","generationLabel":"3","variantLabel":"235B A22B Thinking","createdAt":"2025-09-23T23:04:50.000Z","kind":"multimodal","contextLength":131072,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0.39999999999999997,"completionPerMillion":4,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Qwen3-VL-235B-A22B Thinking is a multimodal model that unifies strong text generation with visual understanding across images and video. The Thinking model is optimized for multimodal reasoning in STEM and math....","huggingFaceId":"Qwen/Qwen3-VL-235B-A22B-Thinking","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3-vl-235b-a22b-thinking","url":"https://openrouter.ai/qwen/qwen3-vl-235b-a22b-thinking","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1401,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 136 · 7,986 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3-vl-30b-a3b-instruct","name":"Qwen3 VL 30B A3B Instruct","releaseLabel":"Qwen3 VL 30B A3B Instruct","generationId":"3","generationLabel":"3","variantLabel":"30B A3B Instruct","createdAt":"2025-10-06T23:47:56.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0.13,"completionPerMillion":0.52,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...","huggingFaceId":"Qwen/Qwen3-VL-30B-A3B-Instruct","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3-vl-30b-a3b-instruct","url":"https://openrouter.ai/qwen/qwen3-vl-30b-a3b-instruct","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3-vl-30b-a3b-thinking","name":"Qwen3 VL 30B A3B Thinking","releaseLabel":"Qwen3 VL 30B A3B Thinking","generationId":"3","generationLabel":"3","variantLabel":"30B A3B Thinking","createdAt":"2025-10-06T23:47:59.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0.19999999999999998,"completionPerMillion":2.4,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Qwen3-VL-30B-A3B-Thinking is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Thinking variant enhances reasoning in STEM, math, and complex tasks. It excels...","huggingFaceId":"Qwen/Qwen3-VL-30B-A3B-Thinking","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3-vl-30b-a3b-thinking","url":"https://openrouter.ai/qwen/qwen3-vl-30b-a3b-thinking","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3-vl-8b-instruct","name":"Qwen3 VL 8B Instruct","releaseLabel":"Qwen3 VL 8B Instruct","generationId":"3","generationLabel":"3","variantLabel":"8B Instruct","createdAt":"2025-10-14T17:35:08.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["image","text"],"outputModalities":["text"],"promptPerMillion":0.117,"completionPerMillion":0.45499999999999996,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Qwen3-VL-8B-Instruct is a multimodal vision-language model from the Qwen3-VL series, built for high-fidelity understanding and reasoning across text, images, and video. It features improved multimodal fusion with Interleaved-MRoPE for long-horizon...","huggingFaceId":"Qwen/Qwen3-VL-8B-Instruct","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3-vl-8b-instruct","url":"https://openrouter.ai/qwen/qwen3-vl-8b-instruct","free":false,"batch":false}],"benchmarks":[{"benchmark":"PBench · average","score":49,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3-VL-8B-Instruct","measuredAt":"2026-05-11","scope":"Registered model evaluation · Community Evals"},{"benchmark":"MDPBench · overall","score":68.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/Delores-Lin/MDPBench","measuredAt":"2026-04-14","scope":"Registered model evaluation · MDPBench leaderboard"},{"benchmark":"MDPBench · digital","score":78.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/Delores-Lin/MDPBench","measuredAt":"2026-04-14","scope":"Registered model evaluation · MDPBench leaderboard"},{"benchmark":"MDPBench · photographed","score":65,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/Delores-Lin/MDPBench","measuredAt":"2026-04-14","scope":"Registered model evaluation · MDPBench leaderboard"},{"benchmark":"MDPBench · latin","score":73.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/Delores-Lin/MDPBench","measuredAt":"2026-04-14","scope":"Registered model evaluation · MDPBench leaderboard"},{"benchmark":"MDPBench · de","score":73.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/Delores-Lin/MDPBench","measuredAt":"2026-04-14","scope":"Registered model evaluation · MDPBench leaderboard"},{"benchmark":"MDPBench · en","score":71.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/Delores-Lin/MDPBench","measuredAt":"2026-04-14","scope":"Registered model evaluation · MDPBench leaderboard"},{"benchmark":"MDPBench · es","score":69.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/Delores-Lin/MDPBench","measuredAt":"2026-04-14","scope":"Registered model evaluation · MDPBench leaderboard"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3-vl-8b-thinking","name":"Qwen3 VL 8B Thinking","releaseLabel":"Qwen3 VL 8B Thinking","generationId":"3","generationLabel":"3","variantLabel":"8B Thinking","createdAt":"2025-10-14T17:42:26.000Z","kind":"multimodal","contextLength":131072,"inputModalities":["image","text"],"outputModalities":["text"],"promptPerMillion":0.18,"completionPerMillion":2.0999999999999996,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Qwen3-VL-8B-Thinking is the reasoning-optimized variant of the Qwen3-VL-8B multimodal model, designed for advanced visual and textual reasoning across complex scenes, documents, and temporal sequences. It integrates enhanced multimodal alignment and...","huggingFaceId":"Qwen/Qwen3-VL-8B-Thinking","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3-vl-8b-thinking","url":"https://openrouter.ai/qwen/qwen3-vl-8b-thinking","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3-vl-32b-instruct","name":"Qwen3 VL 32B Instruct","releaseLabel":"Qwen3 VL 32B Instruct","generationId":"3","generationLabel":"3","variantLabel":"32B Instruct","createdAt":"2025-10-23T14:55:32.000Z","kind":"multimodal","contextLength":131072,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0.10400000000000001,"completionPerMillion":0.41600000000000004,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Qwen3-VL-32B-Instruct is a large-scale multimodal vision-language model designed for high-precision understanding and reasoning across text, images, and video. With 32 billion parameters, it combines deep visual perception with advanced text...","huggingFaceId":"Qwen/Qwen3-VL-32B-Instruct","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3-vl-32b-instruct","url":"https://openrouter.ai/qwen/qwen3-vl-32b-instruct","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]}]},{"id":"qwen-coder","label":"Qwen Coder","models":[{"id":"qwen/qwen-2.5-coder-32b-instruct","name":"Qwen2.5 Coder 32B Instruct","releaseLabel":"Qwen2.5 Coder 32B Instruct","generationId":"2.5","generationLabel":"2.5","variantLabel":"32B Instruct","createdAt":"2024-11-11T23:40:00.000Z","kind":"language","contextLength":32768,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.66,"completionPerMillion":1,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Qwen2.5-Coder is the latest series of Code-Specific Qwen large language models (formerly known as CodeQwen). Qwen2.5-Coder brings the following improvements upon CodeQwen1.5: - Significantly improvements in **code generation**, **code reasoning**...","huggingFaceId":"Qwen/Qwen2.5-Coder-32B-Instruct","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen-2.5-coder-32b-instruct","url":"https://openrouter.ai/qwen/qwen-2.5-coder-32b-instruct","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":46.3,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2024_11_25.csv","measuredAt":"2024-11-25","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1230,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 279 · 5,432 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3-coder-480b-a35b-07-25","name":"Qwen3 Coder 480B A35B","releaseLabel":"Qwen3 Coder 480B A35B","generationId":"3","generationLabel":"3","variantLabel":"Qwen Coder 480B A35B","createdAt":"2025-07-23T00:29:06.000Z","kind":"language","contextLength":262144,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.3,"completionPerMillion":1,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...","huggingFaceId":"Qwen/Qwen3-Coder-480B-A35B-Instruct","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3-coder","url":"https://openrouter.ai/qwen/qwen3-coder","free":false,"batch":false}],"benchmarks":[{"benchmark":"EvasionBench","score":78.2,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"SWE-bench Pro","score":38.7,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"Terminal-Bench","score":23.9,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"SWE-bench_Pro · SWE Bench Pro","score":38.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://scale.com/leaderboard/swe_bench_pro_public","measuredAt":"2026-02-28","scope":"Registered model evaluation · SWE-Bench Pro official evaluation results"},{"benchmark":"terminal-bench-2.0 · terminalbench 2","score":23.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://www.tbench.ai/leaderboard/terminal-bench/2.0","measuredAt":"2025-11-01","scope":"Registered model evaluation · Terminal-Bench Leaderboard"},{"benchmark":"EvasionBench · evasion bench","score":78.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://arxiv.org/abs/2601.09142","measuredAt":"2026-02-10","scope":"Registered model evaluation · EvasionBench Paper"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3-coder-30b-a3b-instruct","name":"Qwen3 Coder 30B A3B Instruct","releaseLabel":"Qwen3 Coder 30B A3B Instruct","generationId":"3","generationLabel":"3","variantLabel":"30B A3B Instruct","createdAt":"2025-07-31T14:32:59.000Z","kind":"language","contextLength":262144,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.07,"completionPerMillion":0.28,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...","huggingFaceId":"Qwen/Qwen3-Coder-30B-A3B-Instruct","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3-coder-30b-a3b-instruct","url":"https://openrouter.ai/qwen/qwen3-coder-30b-a3b-instruct","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3-coder-flash","name":"Qwen3 Coder Flash","releaseLabel":"Qwen3 Coder Flash","generationId":"3","generationLabel":"3","variantLabel":"Flash","createdAt":"2025-09-17T13:25:36.000Z","kind":"language","contextLength":1000000,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.195,"completionPerMillion":0.975,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Qwen3 Coder Flash is Alibaba's fast and cost efficient version of their proprietary Qwen3 Coder Plus. It is a powerful coding agent model specializing in autonomous programming via tool calling...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3-coder-flash","url":"https://openrouter.ai/qwen/qwen3-coder-flash","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3-coder-plus","name":"Qwen3 Coder Plus","releaseLabel":"Qwen3 Coder Plus","generationId":"3","generationLabel":"3","variantLabel":"Plus","createdAt":"2025-09-23T21:25:07.000Z","kind":"language","contextLength":1000000,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.65,"completionPerMillion":3.25,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Qwen3 Coder Plus is Alibaba's proprietary version of the Open Source Qwen3 Coder 480B A35B. It is a powerful coding agent model specializing in autonomous programming via tool calling and...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3-coder-plus","url":"https://openrouter.ai/qwen/qwen3-coder-plus","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3-coder-next-2025-02-03","name":"Qwen3 Coder Next","releaseLabel":"Qwen3 Coder Next","generationId":"3","generationLabel":"3","variantLabel":"Next","createdAt":"2026-02-04T00:15:01.000Z","kind":"language","contextLength":262144,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.12,"completionPerMillion":0.7999999999999999,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...","huggingFaceId":"Qwen/Qwen3-Coder-Next","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3-coder-next","url":"https://openrouter.ai/qwen/qwen3-coder-next","free":false,"batch":false}],"benchmarks":[{"benchmark":"SWE-bench Pro","score":44.3,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"SWE-bench Verified","score":70.6,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"Terminal-Bench","score":36.2,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"SWE-bench_Pro · SWE Bench Pro","score":44.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/papers/2603.00729","measuredAt":"2026-03-16","scope":"Registered model evaluation · Qwen3-Coder-Next technical report"},{"benchmark":"SWE-bench_Verified · swe bench % resolved","score":70.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/papers/2603.00729","measuredAt":"2026-03-16","scope":"Registered model evaluation · Qwen3-Coder-Next technical report"},{"benchmark":"terminal-bench-2.0 · terminalbench 2","score":36.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/Qwen/Qwen3-Coder-Next","measuredAt":"2026-03-16","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]}]},{"id":"qwen-asr","label":"Qwen ASR","models":[{"id":"qwen/qwen3-asr-flash-2026-02-10","name":"Qwen3 ASR Flash","releaseLabel":"Qwen3 ASR Flash","generationId":"3","generationLabel":"3","variantLabel":"Flash · 2026-02-10","createdAt":"2026-05-14T04:26:16.000Z","kind":"audio","contextLength":0,"inputModalities":["audio"],"outputModalities":["transcription"],"promptPerMillion":35,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Qwen3-ASR-Flash is Alibaba's automatic speech recognition service, built on the Qwen3-Omni foundation and trained on tens of millions of hours of multimodal speech data. The model handles 11 languages —...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3-asr-flash-2026-02-10","url":"https://openrouter.ai/qwen/qwen3-asr-flash-2026-02-10","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3-asr-0.6b-20260813","name":"Qwen3 ASR 0.6B","releaseLabel":"Qwen3 ASR 0.6B","generationId":"3","generationLabel":"3","variantLabel":"0.6B","createdAt":"2026-08-13T03:30:33.000Z","kind":"audio","contextLength":0,"inputModalities":["audio"],"outputModalities":["transcription"],"promptPerMillion":3.33,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Qwen3 ASR 0.6B is a compact automatic speech recognition model from Qwen. It supports multilingual language identification and transcription across 30 languages and 22 Chinese dialects, with streaming and offline...","huggingFaceId":"Qwen/Qwen3-ASR-0.6B","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3-asr-0.6b","url":"https://openrouter.ai/qwen/qwen3-asr-0.6b","free":false,"batch":false}],"benchmarks":[{"benchmark":"open-asr-leaderboard · mean wer","score":6.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2026-01-28","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · ami wer","score":11.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2026-01-28","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · earnings22 wer","score":11.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2026-01-28","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · gigaspeech wer","score":9.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2026-01-28","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · librispeech clean wer","score":2.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2026-01-28","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · librispeech other wer","score":4.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2026-01-28","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · spgispeech wer","score":3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2026-01-28","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · tedlium wer","score":2.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2026-01-28","scope":"Registered model evaluation · open-asr-leaderboard"}],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3-asr-1.7b-20260813","name":"Qwen3 ASR 1.7B","releaseLabel":"Qwen3 ASR 1.7B","generationId":"3","generationLabel":"3","variantLabel":"1.7B","createdAt":"2026-08-13T03:44:06.000Z","kind":"audio","contextLength":0,"inputModalities":["audio"],"outputModalities":["transcription"],"promptPerMillion":7.5,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Qwen3 ASR 1.7B is an automatic speech recognition model from Qwen. It supports multilingual language identification and transcription across 30 languages and 22 Chinese dialects, with streaming and offline inference...","huggingFaceId":"Qwen/Qwen3-ASR-1.7B","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3-asr-1.7b","url":"https://openrouter.ai/qwen/qwen3-asr-1.7b","free":false,"batch":false}],"benchmarks":[{"benchmark":"open-asr-leaderboard · mean wer","score":5.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2026-01-28","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · ami wer","score":10.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2026-01-28","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · earnings22 wer","score":10.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2026-01-28","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · gigaspeech wer","score":8.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2026-01-28","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · librispeech clean wer","score":1.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2026-01-28","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · librispeech other wer","score":3.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2026-01-28","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · spgispeech wer","score":2.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2026-01-28","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · tedlium wer","score":2.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2026-01-28","scope":"Registered model evaluation · open-asr-leaderboard"}],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]}]},{"id":"qwen-audio","label":"Qwen Audio","models":[{"id":"qwen/qwen-audio-3.0-tts-flash-20260723","name":"Qwen-Audio-3.0-TTS Flash","releaseLabel":"Qwen-Audio-3.0-TTS Flash","generationId":"3.0","generationLabel":"3.0","variantLabel":"Flash","createdAt":"2026-07-23T14:33:27.000Z","kind":"audio","contextLength":0,"inputModalities":["text"],"outputModalities":["speech"],"promptPerMillion":15,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Qwen-Audio-3.0-TTS Flash is Alibaba's fast, cost-efficient text-to-speech model, generating spoken audio from text via the DashScope Speech Synthesizer API.","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"qwen/qwen-audio-3.0-tts-flash","url":"https://openrouter.ai/qwen/qwen-audio-3.0-tts-flash","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen-audio-3.0-tts-plus-20260723","name":"Qwen-Audio-3.0-TTS Plus","releaseLabel":"Qwen-Audio-3.0-TTS Plus","generationId":"3.0","generationLabel":"3.0","variantLabel":"Plus","createdAt":"2026-07-23T14:33:27.000Z","kind":"audio","contextLength":0,"inputModalities":["text"],"outputModalities":["speech"],"promptPerMillion":20,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Qwen-Audio-3.0-TTS Plus is Alibaba's higher-quality text-to-speech model, generating spoken audio from text via the DashScope Speech Synthesizer API.","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"qwen/qwen-audio-3.0-tts-plus","url":"https://openrouter.ai/qwen/qwen-audio-3.0-tts-plus","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]}]},{"id":"qwen-embedding","label":"Qwen Embedding","models":[{"id":"qwen/qwen3-embedding-4b","name":"Qwen3 Embedding 4B","releaseLabel":"Qwen3 Embedding 4B","generationId":"3","generationLabel":"3","variantLabel":"4B","createdAt":"2025-10-28T14:48:42.000Z","kind":"embedding","contextLength":32768,"inputModalities":["text"],"outputModalities":["embeddings"],"promptPerMillion":0.02,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"The Qwen3 Embedding model series is the latest proprietary model of the Qwen family, specifically designed for text embedding and ranking tasks. This series inherits the exceptional multilingual capabilities, long-text...","huggingFaceId":"Qwen/Qwen3-Embedding-4B","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3-embedding-4b","url":"https://openrouter.ai/qwen/qwen3-embedding-4b","free":false,"batch":false}],"benchmarks":[{"benchmark":"MTEB English retrieval","score":68.5,"unit":"%","sourceName":"MTEB","sourceUrl":"https://leaderboard.mteb.org/","measuredAt":"2026-08-27","scope":"Mean retrieval score · leaderboard rank 7"}],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen3-embedding-8b","name":"Qwen3 Embedding 8B","releaseLabel":"Qwen3 Embedding 8B","generationId":"3","generationLabel":"3","variantLabel":"8B","createdAt":"2025-10-28T19:43:42.000Z","kind":"embedding","contextLength":32768,"inputModalities":["text"],"outputModalities":["embeddings"],"promptPerMillion":0.01,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"The Qwen3 Embedding model series is the latest proprietary model of the Qwen family, specifically designed for text embedding and ranking tasks. This series inherits the exceptional multilingual capabilities, long-text...","huggingFaceId":"Qwen/Qwen3-Embedding-8B","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3-embedding-8b","url":"https://openrouter.ai/qwen/qwen3-embedding-8b","free":false,"batch":false}],"benchmarks":[{"benchmark":"MTEB English retrieval","score":69.4,"unit":"%","sourceName":"MTEB","sourceUrl":"https://leaderboard.mteb.org/","measuredAt":"2026-08-27","scope":"Mean retrieval score · leaderboard rank 5"}],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]}]},{"id":"qwen-image","label":"Qwen Image","models":[{"id":"qwen/qwen-image-3-20260805","name":"Qwen Image 3","releaseLabel":"3","generationId":"3","generationLabel":"3","variantLabel":"Base","createdAt":"2026-08-05T01:49:08.000Z","kind":"image","contextLength":65536,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":7.185628742514971,"supportsTools":false,"supportsReasoning":false,"description":"Qwen Image 3 is a unified image generation and editing model from Qwen. It supports precise rendering of text and details as small as 10px, along with a richer world...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"qwen/qwen-image-3","url":"https://openrouter.ai/qwen/qwen-image-3","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]},{"id":"qwen/qwen-image-3-pro-20260805","name":"Qwen Image 3 Pro","releaseLabel":"3 Pro","generationId":"3","generationLabel":"3","variantLabel":"Pro","createdAt":"2026-08-05T01:49:08.000Z","kind":"image","contextLength":65536,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":9.580838323353289,"supportsTools":false,"supportsReasoning":false,"description":"Qwen Image 3 Pro is an image generation and editing model from Qwen. It supports precise rendering of text and details as small as 10px, along with richer world knowledge...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"qwen/qwen-image-3-pro","url":"https://openrouter.ai/qwen/qwen-image-3-pro","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]}]},{"id":"qwen-reranker","label":"Qwen Reranker","models":[{"id":"qwen/qwen3-reranker-8b","name":"Qwen3 Reranker 8B","releaseLabel":"Qwen3 Reranker 8B","generationId":"3","generationLabel":"3","variantLabel":"8B","createdAt":"2026-08-13T05:08:04.000Z","kind":"language","contextLength":40960,"inputModalities":["text"],"outputModalities":["rerank"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Qwen3 Reranker 8B is a text reranking model from Alibaba Cloud built on the Qwen3 architecture. It evaluates query-document pairs to produce relevance scores for use in retrieval and RAG...","huggingFaceId":"Qwen/Qwen3-Reranker-8B","routes":[{"provider":"OpenRouter","modelId":"qwen/qwen3-reranker-8b","url":"https://openrouter.ai/qwen/qwen3-reranker-8b","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"39e3213c-bb0e-4cbd-8b79-3665f218bd90","title":"Controlling Reasoning Effort in LLMs","href":"/intel/39e3213c-bb0e-4cbd-8b79-3665f218bd90","hook":"If you tune reasoning effort in models like GPT-5, Qwen3, or gpt-oss, this deep-dive explains how those low/medium/high modes are actually trained and implem…"},{"id":"22a2b202-6872-468c-ad2b-904abe54c9e0","title":"Understanding and Implementing Qwen3 From Scratch","href":"/intel/22a2b202-6872-468c-ad2b-904abe54c9e0","hook":"If you want to actually understand how modern open-weight LLMs work, this walks you through reimplementing Qwen3's dense and Mixture-of-Experts variants in p…"}]}]}]},{"id":"moonshotai","name":"Moonshot AI","domain":"moonshot.ai","major":true,"modelCount":7,"series":[{"id":"kimi","label":"Kimi","models":[{"id":"moonshotai/kimi-k2","name":"Kimi K2 0711","releaseLabel":"K2 0711","generationId":"releases","generationLabel":"Releases","variantLabel":"K2 0711","createdAt":"2025-07-11T19:47:32.000Z","kind":"language","contextLength":131072,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.5700000000000001,"completionPerMillion":2.3,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Kimi K2 Instruct is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32 billion active per forward pass. It is optimized for...","huggingFaceId":"moonshotai/Kimi-K2-Instruct","routes":[{"provider":"OpenRouter","modelId":"moonshotai/kimi-k2","url":"https://openrouter.ai/moonshotai/kimi-k2","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1371,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 164 · 27,722 votes"},{"benchmark":"SWE-bench Pro","score":27.7,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"Terminal-Bench","score":27.8,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"APEX-SWE · apex-swe","score":11.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://www.mercor.com/apex/apex-swe-leaderboard/","measuredAt":"2026-04-22","scope":"Registered model evaluation · APEX-SWE Leaderboard"},{"benchmark":"SWE-bench_Pro · SWE Bench Pro","score":27.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://scale.com/leaderboard/swe_bench_pro_public","measuredAt":"2026-04-23","scope":"Registered model evaluation · SWE-Bench Pro official evaluation results"},{"benchmark":"terminal-bench-2.0 · terminalbench 2","score":27.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://www.tbench.ai/leaderboard/terminal-bench/2.0","measuredAt":"2025-11-01","scope":"Registered model evaluation · Terminal-Bench Leaderboard"},{"benchmark":"SWE-bench_Multilingual · swe bench multilingual % resolved","score":47.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2-Instruct","measuredAt":"2026-08-10","scope":"Registered model evaluation · Model Card"},{"benchmark":"MMLU-Pro · mmlu pro","score":81,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/4a/ae/4aae2732-e24b-42a2-9dc1-a35cd215a8ec.json","measuredAt":"2026-06-30","scope":"Registered model evaluation · EvalEval"},{"benchmark":"apex-agents · apex-agents","score":4.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://www.mercor.com/apex/apex-agents-leaderboard/","measuredAt":"2026-04-22","scope":"Registered model evaluation · APEX-Agents Leaderboard"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"a7fd9d16-e7b2-45fd-bf93-f8f8b5b2c48b","title":"Running Kimi K3 on my desk? It will require 4 x 512GB M3 Ultra Mac Studios (2TB ","href":"/intel/a7fd9d16-e7b2-45fd-bf93-f8f8b5b2c48b","hook":"If you are sizing hardware for local frontier-scale MoE models, this gives ballpark memory, bandwidth, and throughput figures plus the interconnect technique…"},{"id":"594732dc-64a7-4fae-86e2-9d7b00f2ea59","title":"Kimi K3: The open-weights escalation","href":"/intel/594732dc-64a7-4fae-86e2-9d7b00f2ea59","hook":"What Kimi K3's release signals about the open-weights race and the capability gap practitioners should plan around."}]},{"id":"moonshotai/kimi-k2-0905","name":"Kimi K2 0905","releaseLabel":"K2 0905","generationId":"0905","generationLabel":"0905","variantLabel":"K2","createdAt":"2025-09-04T21:25:47.000Z","kind":"language","contextLength":262144,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.6,"completionPerMillion":2.5,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Kimi K2 0905 is the September update of [Kimi K2 0711](moonshotai/kimi-k2). It is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32...","huggingFaceId":"moonshotai/Kimi-K2-Instruct-0905","routes":[{"provider":"OpenRouter","modelId":"moonshotai/kimi-k2-0905","url":"https://openrouter.ai/moonshotai/kimi-k2-0905","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1379,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 154 · 11,807 votes"},{"benchmark":"EvasionBench","score":66.7,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"EvasionBench · evasion bench","score":66.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://arxiv.org/abs/2601.09142","measuredAt":"2026-02-10","scope":"Registered model evaluation · EvasionBench Paper"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"a7fd9d16-e7b2-45fd-bf93-f8f8b5b2c48b","title":"Running Kimi K3 on my desk? It will require 4 x 512GB M3 Ultra Mac Studios (2TB ","href":"/intel/a7fd9d16-e7b2-45fd-bf93-f8f8b5b2c48b","hook":"If you are sizing hardware for local frontier-scale MoE models, this gives ballpark memory, bandwidth, and throughput figures plus the interconnect technique…"},{"id":"594732dc-64a7-4fae-86e2-9d7b00f2ea59","title":"Kimi K3: The open-weights escalation","href":"/intel/594732dc-64a7-4fae-86e2-9d7b00f2ea59","hook":"What Kimi K3's release signals about the open-weights race and the capability gap practitioners should plan around."}]},{"id":"moonshotai/kimi-k2-thinking-20251106","name":"Kimi K2 Thinking","releaseLabel":"K2 Thinking","generationId":"releases","generationLabel":"Releases","variantLabel":"K2 Thinking","createdAt":"2025-11-06T14:50:22.000Z","kind":"language","contextLength":262144,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.6,"completionPerMillion":2.5,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Kimi K2 Thinking is Moonshot AI’s most advanced open reasoning model to date, extending the K2 series into agentic, long-horizon reasoning. Built on the trillion-parameter Mixture-of-Experts (MoE) architecture introduced in...","huggingFaceId":"moonshotai/Kimi-K2-Thinking","routes":[{"provider":"OpenRouter","modelId":"moonshotai/kimi-k2-thinking","url":"https://openrouter.ai/moonshotai/kimi-k2-thinking","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":62.3,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1414,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 117 · 62,107 votes"},{"benchmark":"GPQA","score":84.5,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"Humanity's Last Exam","score":23.9,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"MMLU-Pro","score":84.6,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"SWE-bench Verified","score":71.3,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"Terminal-Bench","score":35.7,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"SWE-bench_Verified · swe bench % resolved","score":71.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2-Thinking","measuredAt":"2026-03-17","scope":"Registered model evaluation · Model Card"},{"benchmark":"terminal-bench-2.0 · terminalbench 2","score":35.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://www.tbench.ai/leaderboard/terminal-bench/2.0","measuredAt":"2025-11-11","scope":"Registered model evaluation · Terminal-Bench Leaderboard"},{"benchmark":"hle · hle","score":44.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2-Thinking","measuredAt":"2026-02-03","scope":"Registered model evaluation · Model Card"},{"benchmark":"gpqa · diamond","score":84.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2-Thinking","measuredAt":"2026-01-27","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"a7fd9d16-e7b2-45fd-bf93-f8f8b5b2c48b","title":"Running Kimi K3 on my desk? It will require 4 x 512GB M3 Ultra Mac Studios (2TB ","href":"/intel/a7fd9d16-e7b2-45fd-bf93-f8f8b5b2c48b","hook":"If you are sizing hardware for local frontier-scale MoE models, this gives ballpark memory, bandwidth, and throughput figures plus the interconnect technique…"},{"id":"594732dc-64a7-4fae-86e2-9d7b00f2ea59","title":"Kimi K3: The open-weights escalation","href":"/intel/594732dc-64a7-4fae-86e2-9d7b00f2ea59","hook":"What Kimi K3's release signals about the open-weights race and the capability gap practitioners should plan around."}]},{"id":"moonshotai/kimi-k2.5-0127","name":"Kimi K2.5","releaseLabel":"K2.5","generationId":"releases","generationLabel":"Releases","variantLabel":"K2.5","createdAt":"2026-01-27T04:11:16.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0.6,"completionPerMillion":3,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...","huggingFaceId":"moonshotai/Kimi-K2.5","routes":[{"provider":"OpenRouter","modelId":"moonshotai/kimi-k2.5","url":"https://openrouter.ai/moonshotai/kimi-k2.5","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":69.2,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1445,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 49 · 70,943 votes"},{"benchmark":"AIME 2026","score":95.8,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"GPQA","score":87.6,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"Humanity's Last Exam","score":50.2,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"HMMT 2026","score":87.1,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"MMLU-Pro","score":87.1,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"SWE-bench Pro","score":50.7,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"SWE-bench Verified","score":70.8,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"Terminal-Bench","score":43.2,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"hle · hle","score":50.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.5","measuredAt":"2026-02-04","scope":"Registered model evaluation · Model Card"},{"benchmark":"SWE-bench_Multilingual · swe bench multilingual % resolved","score":73,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.5","measuredAt":"2026-08-10","scope":"Registered model evaluation · Model Card"},{"benchmark":"Real5-OmniDocBench · overall","score":89.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/PaddlePaddle/Real5-OmniDocBench","measuredAt":"2026-08-08","scope":"Registered model evaluation · Real5-OmniDocBench Leaderboard"},{"benchmark":"Real5-OmniDocBench · scanning","score":89.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/PaddlePaddle/Real5-OmniDocBench","measuredAt":"2026-08-08","scope":"Registered model evaluation · Real5-OmniDocBench Leaderboard"},{"benchmark":"Real5-OmniDocBench · warping","score":88.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/PaddlePaddle/Real5-OmniDocBench","measuredAt":"2026-08-08","scope":"Registered model evaluation · Real5-OmniDocBench Leaderboard"},{"benchmark":"Real5-OmniDocBench · screen photography","score":88.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/PaddlePaddle/Real5-OmniDocBench","measuredAt":"2026-08-08","scope":"Registered model evaluation · Real5-OmniDocBench Leaderboard"},{"benchmark":"Real5-OmniDocBench · illumination","score":89.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/PaddlePaddle/Real5-OmniDocBench","measuredAt":"2026-08-08","scope":"Registered model evaluation · Real5-OmniDocBench Leaderboard"},{"benchmark":"Real5-OmniDocBench · skew","score":88.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/PaddlePaddle/Real5-OmniDocBench","measuredAt":"2026-08-08","scope":"Registered model evaluation · Real5-OmniDocBench Leaderboard"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"a7fd9d16-e7b2-45fd-bf93-f8f8b5b2c48b","title":"Running Kimi K3 on my desk? It will require 4 x 512GB M3 Ultra Mac Studios (2TB ","href":"/intel/a7fd9d16-e7b2-45fd-bf93-f8f8b5b2c48b","hook":"If you are sizing hardware for local frontier-scale MoE models, this gives ballpark memory, bandwidth, and throughput figures plus the interconnect technique…"},{"id":"594732dc-64a7-4fae-86e2-9d7b00f2ea59","title":"Kimi K3: The open-weights escalation","href":"/intel/594732dc-64a7-4fae-86e2-9d7b00f2ea59","hook":"What Kimi K3's release signals about the open-weights race and the capability gap practitioners should plan around."}]},{"id":"moonshotai/kimi-k2.6-20260420","name":"Kimi K2.6","releaseLabel":"K2.6","generationId":"releases","generationLabel":"Releases","variantLabel":"K2.6","createdAt":"2026-04-20T15:36:42.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0.95,"completionPerMillion":4,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...","huggingFaceId":"moonshotai/Kimi-K2.6","routes":[{"provider":"OpenRouter","modelId":"moonshotai/kimi-k2.6","url":"https://openrouter.ai/moonshotai/kimi-k2.6","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":70.9,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1455,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 34 · 37,543 votes"},{"benchmark":"aime_2026 · MathArena/aime 2026","score":96.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6","measuredAt":"2026-04-20","scope":"Registered model evaluation · Model Card"},{"benchmark":"gpqa · diamond","score":90.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6","measuredAt":"2026-04-20","scope":"Registered model evaluation · Model Card"},{"benchmark":"hle · hle","score":34.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6","measuredAt":"2026-04-20","scope":"Registered model evaluation · Model Card"},{"benchmark":"hmmt_feb_2026 · MathArena/hmmt feb 2026","score":92.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6","measuredAt":"2026-04-20","scope":"Registered model evaluation · Model Card"},{"benchmark":"MMMU_Pro · mmmu pro vision","score":79.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6","measuredAt":"2026-04-20","scope":"Registered model evaluation · Model Card"},{"benchmark":"SWE-bench_Pro · SWE Bench Pro","score":58.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6","measuredAt":"2026-04-20","scope":"Registered model evaluation · Model Card"},{"benchmark":"SWE-bench_Verified · swe bench % resolved","score":80.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6","measuredAt":"2026-04-20","scope":"Registered model evaluation · Model Card"},{"benchmark":"terminal-bench-2.0 · terminalbench 2","score":66.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6","measuredAt":"2026-04-20","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"a7fd9d16-e7b2-45fd-bf93-f8f8b5b2c48b","title":"Running Kimi K3 on my desk? It will require 4 x 512GB M3 Ultra Mac Studios (2TB ","href":"/intel/a7fd9d16-e7b2-45fd-bf93-f8f8b5b2c48b","hook":"If you are sizing hardware for local frontier-scale MoE models, this gives ballpark memory, bandwidth, and throughput figures plus the interconnect technique…"},{"id":"594732dc-64a7-4fae-86e2-9d7b00f2ea59","title":"Kimi K3: The open-weights escalation","href":"/intel/594732dc-64a7-4fae-86e2-9d7b00f2ea59","hook":"What Kimi K3's release signals about the open-weights race and the capability gap practitioners should plan around."}]},{"id":"moonshotai/kimi-k2.7-code-20260612","name":"Kimi K2.7 Code","releaseLabel":"K2.7 Code","generationId":"releases","generationLabel":"Releases","variantLabel":"K2.7 Code","createdAt":"2026-06-12T12:12:41.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0.66,"completionPerMillion":3.4,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","huggingFaceId":"moonshotai/Kimi-K2.7-Code","routes":[{"provider":"OpenRouter","modelId":"moonshotai/kimi-k2.7-code","url":"https://openrouter.ai/moonshotai/kimi-k2.7-code","free":false,"batch":false},{"provider":"OpenRouter","modelId":"moonshotai/kimi-k2.7-code:batch","url":"https://openrouter.ai/moonshotai/kimi-k2.7-code:batch","free":false,"batch":true}],"benchmarks":[{"benchmark":"LiveBench overall","score":68.8,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"WildClawBench · overall","score":46.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://internlm.github.io/WildClawBench","measuredAt":"2026-07-30","scope":"Registered model evaluation · WildClawBench"},{"benchmark":"Long-Horizon-Terminal-Bench · lhtb solved","score":3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://zli12321.github.io/LHTB/leaderboard.html","measuredAt":"2026-07-16","scope":"Registered model evaluation · LHTB leaderboard"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"a7fd9d16-e7b2-45fd-bf93-f8f8b5b2c48b","title":"Running Kimi K3 on my desk? It will require 4 x 512GB M3 Ultra Mac Studios (2TB ","href":"/intel/a7fd9d16-e7b2-45fd-bf93-f8f8b5b2c48b","hook":"If you are sizing hardware for local frontier-scale MoE models, this gives ballpark memory, bandwidth, and throughput figures plus the interconnect technique…"},{"id":"594732dc-64a7-4fae-86e2-9d7b00f2ea59","title":"Kimi K3: The open-weights escalation","href":"/intel/594732dc-64a7-4fae-86e2-9d7b00f2ea59","hook":"What Kimi K3's release signals about the open-weights race and the capability gap practitioners should plan around."}]},{"id":"moonshotai/kimi-k3-20260715","name":"Kimi K3","releaseLabel":"K3","generationId":"releases","generationLabel":"Releases","variantLabel":"K3","createdAt":"2026-07-16T15:30:58.000Z","kind":"multimodal","contextLength":1048576,"inputModalities":["text","image","video"],"outputModalities":["text"],"promptPerMillion":3,"completionPerMillion":15,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","huggingFaceId":"moonshotai/Kimi-K3","routes":[{"provider":"OpenRouter","modelId":"moonshotai/kimi-k3","url":"https://openrouter.ai/moonshotai/kimi-k3","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":79.5,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"AA Intelligence Index","score":59.7,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/kimi-k3","measuredAt":"2026-08-27","scope":"Current public index; highest tested effort configuration"},{"benchmark":"AA Agentic Index","score":54.3,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/kimi-k3","measuredAt":"2026-08-27","scope":"Current public agentic composite"},{"benchmark":"SciCode","score":58.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/kimi-k3","measuredAt":"2026-08-27","scope":"Scientific coding evaluation"},{"benchmark":"AA Omniscience Index","score":19.7,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/kimi-k3","measuredAt":"2026-08-27","scope":"Current public knowledge and hallucination index"},{"benchmark":"MLCR overall","score":38.3,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/kimi-k3","measuredAt":"2026-08-27","scope":"Machine-learning code reasoning evaluation"},{"benchmark":"GDPval-AA normalized","score":58.5,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/kimi-k3","measuredAt":"2026-08-27","scope":"Normalized real-world knowledge-work evaluation"},{"benchmark":"ITBench SRE","score":47.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/kimi-k3","measuredAt":"2026-08-27","scope":"Site-reliability engineering agent tasks"},{"benchmark":"τ²-bench Banking","score":46,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/kimi-k3","measuredAt":"2026-08-27","scope":"Tool-agent benchmark in the banking domain"},{"benchmark":"Terminal-Bench 2.1","score":85,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/kimi-k3","measuredAt":"2026-08-27","scope":"Terminal-agent task completion"},{"benchmark":"LCR","score":82.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/kimi-k3","measuredAt":"2026-08-27","scope":"Long-context reasoning evaluation"},{"benchmark":"Humanity's Last Exam","score":46.9,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/kimi-k3","measuredAt":"2026-08-27","scope":"Broad expert-level knowledge evaluation"},{"benchmark":"GPQA","score":93.5,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/kimi-k3","measuredAt":"2026-08-27","scope":"Graduate-level science questions"},{"benchmark":"CritPt","score":23.4,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/kimi-k3","measuredAt":"2026-08-27","scope":"Critical-point reasoning evaluation"},{"benchmark":"APEX Agents","score":41.3,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/kimi-k3","measuredAt":"2026-08-27","scope":"Professional agent task evaluation"},{"benchmark":"MMMU Pro","score":80.5,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/kimi-k3","measuredAt":"2026-08-27","scope":"Multimodal expert reasoning evaluation"},{"benchmark":"Analyst Agent","score":38.8,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/kimi-k3","measuredAt":"2026-08-27","scope":"Analyst workflow agent tasks"},{"benchmark":"Enterprise Ops Gym","score":45.3,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/kimi-k3","measuredAt":"2026-08-27","scope":"Enterprise operations agent tasks"},{"benchmark":"apex-agents · apex-agents","score":41,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3","measuredAt":"2026-07-27","scope":"Registered model evaluation · Model Card"},{"benchmark":"deep-swe · deep swe","score":67.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3","measuredAt":"2026-07-27","scope":"Registered model evaluation · Model Card"},{"benchmark":"gpqa · diamond","score":93.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3","measuredAt":"2026-07-27","scope":"Registered model evaluation · Model Card"},{"benchmark":"hle · hle","score":56,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3","measuredAt":"2026-07-27","scope":"Registered model evaluation · Model Card"},{"benchmark":"ExtractBench · mean","score":83.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/llamaindex/ExtractBench","measuredAt":"2026-08-25","scope":"Registered model evaluation · ExtractBench"},{"benchmark":"ExtractBench · short","score":94.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/llamaindex/ExtractBench","measuredAt":"2026-08-25","scope":"Registered model evaluation · ExtractBench"},{"benchmark":"ExtractBench · medium","score":69.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/llamaindex/ExtractBench","measuredAt":"2026-08-25","scope":"Registered model evaluation · ExtractBench"},{"benchmark":"ExtractBench · long","score":5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/llamaindex/ExtractBench","measuredAt":"2026-08-25","scope":"Registered model evaluation · ExtractBench"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"a7fd9d16-e7b2-45fd-bf93-f8f8b5b2c48b","title":"Running Kimi K3 on my desk? It will require 4 x 512GB M3 Ultra Mac Studios (2TB ","href":"/intel/a7fd9d16-e7b2-45fd-bf93-f8f8b5b2c48b","hook":"If you are sizing hardware for local frontier-scale MoE models, this gives ballpark memory, bandwidth, and throughput figures plus the interconnect technique…"},{"id":"594732dc-64a7-4fae-86e2-9d7b00f2ea59","title":"Kimi K3: The open-weights escalation","href":"/intel/594732dc-64a7-4fae-86e2-9d7b00f2ea59","hook":"What Kimi K3's release signals about the open-weights race and the capability gap practitioners should plan around."}]}]}]},{"id":"meta","name":"Meta","domain":"ai.meta.com","major":true,"modelCount":13,"series":[{"id":"llama","label":"Llama","models":[{"id":"meta/llama-3.1-70b-instruct","name":"Llama 3.1 70B Instruct","releaseLabel":"3.1 70B Instruct","generationId":"3.1","generationLabel":"3.1","variantLabel":"70B Instruct","createdAt":"2024-07-23T00:00:00.000Z","kind":"language","contextLength":131072,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.39999999999999997,"completionPerMillion":0.39999999999999997,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 70B instruct-tuned version is optimized for high quality dialogue usecases. It has demonstrated strong...","huggingFaceId":"meta-llama/Meta-Llama-3.1-70B-Instruct","routes":[{"provider":"OpenRouter","modelId":"meta-llama/llama-3.1-70b-instruct","url":"https://openrouter.ai/meta-llama/llama-3.1-70b-instruct","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1261,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 268 · 55,240 votes"},{"benchmark":"MMLU-Pro · mmlu pro","score":62.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/7f/49/7f49effc-1188-4682-bfc2-d6ea176cfffa.json","measuredAt":"2026-06-30","scope":"Registered model evaluation · EvalEval"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"6f948657-6919-4b4d-bcd3-3a3a79839c6d","title":"Mark Zuckerberg — AI will write most Meta code in 18 months","href":"/intel/6f948657-6919-4b4d-bcd3-3a3a79839c6d","hook":"A frontier-lab CEO's candid take on Llama 4, benchmark gaming, open-source model economics, and how fast AI will take over code generation — useful signal fo…"},{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"dd3c95ad-0fac-49bd-9582-e0c78686e14f","title":"AutoAgent: Self-Optimizing AI Agents","href":"/intel/dd3c95ad-0fac-49bd-9582-e0c78686e14f","hook":"An open-source meta-agent that tunes task agents — prompts, tools, orchestration — until performance climbs, no human in the loop. A working reference if you…"}]},{"id":"meta/llama-3.1-8b-instruct","name":"Llama 3.1 8B Instruct","releaseLabel":"3.1 8B Instruct","generationId":"3.1","generationLabel":"3.1","variantLabel":"8B Instruct","createdAt":"2024-07-23T00:00:00.000Z","kind":"language","contextLength":131072,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.049999999999999996,"completionPerMillion":0.08,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...","huggingFaceId":"meta-llama/Meta-Llama-3.1-8B-Instruct","routes":[{"provider":"OpenRouter","modelId":"meta-llama/llama-3.1-8b-instruct","url":"https://openrouter.ai/meta-llama/llama-3.1-8b-instruct","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1187,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 305 · 49,605 votes"},{"benchmark":"GPQA","score":30.4,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"GSM8K","score":84.5,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"MMLU-Pro","score":48.3,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"signalbench · src","score":0.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation · signalbench raw per-item responses"},{"benchmark":"signalbench · time","score":0.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation"},{"benchmark":"signalbench · access deny","score":0.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation"},{"benchmark":"signalbench · memory label","score":0.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation"},{"benchmark":"signalbench · injection","score":0.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation"},{"benchmark":"signalbench · bot policy","score":0.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation"},{"benchmark":"LEXam · open question","score":10,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://lexam-benchmark.github.io/","measuredAt":"2026-06-02","scope":"Registered model evaluation · LEXam Leaderboard"},{"benchmark":"LEXam · mcq 4 choices","score":24,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://lexam-benchmark.github.io/","measuredAt":"2026-06-02","scope":"Registered model evaluation · LEXam Leaderboard"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"6f948657-6919-4b4d-bcd3-3a3a79839c6d","title":"Mark Zuckerberg — AI will write most Meta code in 18 months","href":"/intel/6f948657-6919-4b4d-bcd3-3a3a79839c6d","hook":"A frontier-lab CEO's candid take on Llama 4, benchmark gaming, open-source model economics, and how fast AI will take over code generation — useful signal fo…"},{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"dd3c95ad-0fac-49bd-9582-e0c78686e14f","title":"AutoAgent: Self-Optimizing AI Agents","href":"/intel/dd3c95ad-0fac-49bd-9582-e0c78686e14f","hook":"An open-source meta-agent that tunes task agents — prompts, tools, orchestration — until performance climbs, no human in the loop. A working reference if you…"}]},{"id":"meta/llama-3.2-1b-instruct","name":"Llama 3.2 1B Instruct","releaseLabel":"3.2 1B Instruct","generationId":"3.2","generationLabel":"3.2","variantLabel":"1B Instruct","createdAt":"2024-09-25T00:00:00.000Z","kind":"language","contextLength":60000,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.027,"completionPerMillion":0.201,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Llama 3.2 1B is a 1-billion-parameter language model focused on efficiently performing natural language tasks, such as summarization, dialogue, and multilingual text analysis. Its smaller size allows it to operate...","huggingFaceId":"meta-llama/Llama-3.2-1B-Instruct","routes":[{"provider":"OpenRouter","modelId":"meta-llama/llama-3.2-1b-instruct","url":"https://openrouter.ai/meta-llama/llama-3.2-1b-instruct","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1055,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 367 · 8,045 votes"},{"benchmark":"gpqa · diamond","score":18.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/aa/ad/aaad0db3-7c5b-4834-a1a2-d96624bdd003.json","measuredAt":"2026-04-16","scope":"Registered model evaluation · EvalEval"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"6f948657-6919-4b4d-bcd3-3a3a79839c6d","title":"Mark Zuckerberg — AI will write most Meta code in 18 months","href":"/intel/6f948657-6919-4b4d-bcd3-3a3a79839c6d","hook":"A frontier-lab CEO's candid take on Llama 4, benchmark gaming, open-source model economics, and how fast AI will take over code generation — useful signal fo…"},{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"dd3c95ad-0fac-49bd-9582-e0c78686e14f","title":"AutoAgent: Self-Optimizing AI Agents","href":"/intel/dd3c95ad-0fac-49bd-9582-e0c78686e14f","hook":"An open-source meta-agent that tunes task agents — prompts, tools, orchestration — until performance climbs, no human in the loop. A working reference if you…"}]},{"id":"meta/llama-3.2-3b-instruct","name":"Llama 3.2 3B Instruct","releaseLabel":"3.2 3B Instruct","generationId":"3.2","generationLabel":"3.2","variantLabel":"3B Instruct","createdAt":"2024-09-25T00:00:00.000Z","kind":"language","contextLength":131072,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.049999999999999996,"completionPerMillion":0.33,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Llama 3.2 3B is a 3-billion-parameter multilingual large language model, optimized for advanced natural language processing tasks like dialogue generation, reasoning, and summarization. Designed with the latest transformer architecture, it...","huggingFaceId":"meta-llama/Llama-3.2-3B-Instruct","routes":[{"provider":"OpenRouter","modelId":"meta-llama/llama-3.2-3b-instruct","url":"https://openrouter.ai/meta-llama/llama-3.2-3b-instruct","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1110,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 339 · 7,936 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"6f948657-6919-4b4d-bcd3-3a3a79839c6d","title":"Mark Zuckerberg — AI will write most Meta code in 18 months","href":"/intel/6f948657-6919-4b4d-bcd3-3a3a79839c6d","hook":"A frontier-lab CEO's candid take on Llama 4, benchmark gaming, open-source model economics, and how fast AI will take over code generation — useful signal fo…"},{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"dd3c95ad-0fac-49bd-9582-e0c78686e14f","title":"AutoAgent: Self-Optimizing AI Agents","href":"/intel/dd3c95ad-0fac-49bd-9582-e0c78686e14f","hook":"An open-source meta-agent that tunes task agents — prompts, tools, orchestration — until performance climbs, no human in the loop. A working reference if you…"}]},{"id":"meta/llama-3.3-70b-instruct","name":"Llama 3.3 70B Instruct","releaseLabel":"3.3 70B Instruct","generationId":"3.3","generationLabel":"3.3","variantLabel":"70B Instruct","createdAt":"2024-12-06T17:28:57.000Z","kind":"language","contextLength":131072,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.71,"completionPerMillion":0.71,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...","huggingFaceId":"meta-llama/Llama-3.3-70B-Instruct","routes":[{"provider":"OpenRouter","modelId":"meta-llama/llama-3.3-70b-instruct","url":"https://openrouter.ai/meta-llama/llama-3.3-70b-instruct","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1275,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 254 · 54,774 votes"},{"benchmark":"signalbench · src","score":0.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation · signalbench raw per-item responses"},{"benchmark":"signalbench · time","score":0.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation"},{"benchmark":"signalbench · access deny","score":0.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation"},{"benchmark":"signalbench · memory label","score":0.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation"},{"benchmark":"signalbench · injection","score":0.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation"},{"benchmark":"signalbench · bot policy","score":0.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation"},{"benchmark":"gpqa · diamond","score":51.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/44/8b/448b1084-4794-44f8-b895-d76737e71c5d.json","measuredAt":"2026-04-16","scope":"Registered model evaluation · EvalEval"},{"benchmark":"gsm8k · gsm8k","score":94.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/84/2f/842f32fc-84e3-4dd7-b9ce-8cadeb908e16.json","measuredAt":"2025-03-19","scope":"Registered model evaluation · EvalEval"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"6f948657-6919-4b4d-bcd3-3a3a79839c6d","title":"Mark Zuckerberg — AI will write most Meta code in 18 months","href":"/intel/6f948657-6919-4b4d-bcd3-3a3a79839c6d","hook":"A frontier-lab CEO's candid take on Llama 4, benchmark gaming, open-source model economics, and how fast AI will take over code generation — useful signal fo…"},{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"dd3c95ad-0fac-49bd-9582-e0c78686e14f","title":"AutoAgent: Self-Optimizing AI Agents","href":"/intel/dd3c95ad-0fac-49bd-9582-e0c78686e14f","hook":"An open-source meta-agent that tunes task agents — prompts, tools, orchestration — until performance climbs, no human in the loop. A working reference if you…"}]},{"id":"meta/llama-4-scout-17b-16e-instruct","name":"Llama 4 Scout","releaseLabel":"4 Scout","generationId":"4","generationLabel":"4","variantLabel":"Scout","createdAt":"2025-04-05T19:31:59.000Z","kind":"multimodal","contextLength":1310720,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0.11,"completionPerMillion":0.33999999999999997,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Llama 4 Scout 17B Instruct (16E) is a mixture-of-experts (MoE) language model developed by Meta, activating 17 billion parameters out of a total of 109B. It supports native multimodal input...","huggingFaceId":"meta-llama/Llama-4-Scout-17B-16E-Instruct","routes":[{"provider":"OpenRouter","modelId":"meta-llama/llama-4-scout","url":"https://openrouter.ai/meta-llama/llama-4-scout","free":false,"batch":false}],"benchmarks":[{"benchmark":"gpqa · diamond","score":46.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/4e/b4/4eb49217-c927-406a-b3e2-121a5459c72d.json","measuredAt":"2026-04-16","scope":"Registered model evaluation · EvalEval"},{"benchmark":"MMLU-Pro · mmlu pro","score":74.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/fa/e1/fae13a5b-eda7-4a54-8f64-fb166e4a39b8.json","measuredAt":"2026-06-30","scope":"Registered model evaluation · EvalEval"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"6f948657-6919-4b4d-bcd3-3a3a79839c6d","title":"Mark Zuckerberg — AI will write most Meta code in 18 months","href":"/intel/6f948657-6919-4b4d-bcd3-3a3a79839c6d","hook":"A frontier-lab CEO's candid take on Llama 4, benchmark gaming, open-source model economics, and how fast AI will take over code generation — useful signal fo…"},{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"dd3c95ad-0fac-49bd-9582-e0c78686e14f","title":"AutoAgent: Self-Optimizing AI Agents","href":"/intel/dd3c95ad-0fac-49bd-9582-e0c78686e14f","hook":"An open-source meta-agent that tunes task agents — prompts, tools, orchestration — until performance climbs, no human in the loop. A working reference if you…"}]},{"id":"meta/llama-4-maverick-17b-128e-instruct","name":"Llama 4 Maverick","releaseLabel":"4 Maverick","generationId":"4","generationLabel":"4","variantLabel":"Maverick","createdAt":"2025-04-05T19:37:02.000Z","kind":"multimodal","contextLength":1048576,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0.19999999999999998,"completionPerMillion":0.7999999999999999,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Llama 4 Maverick 17B Instruct (128E) is a high-capacity multimodal language model from Meta, built on a mixture-of-experts (MoE) architecture with 128 experts and 17 billion active parameters per forward...","huggingFaceId":"meta-llama/Llama-4-Maverick-17B-128E-Instruct","routes":[{"provider":"OpenRouter","modelId":"meta-llama/llama-4-maverick","url":"https://openrouter.ai/meta-llama/llama-4-maverick","free":false,"batch":false}],"benchmarks":[{"benchmark":"SWE-bench Pro","score":5.2,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"hle · none","score":5.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/81/5e/815ee945-0f06-4dbb-8589-9953152ae291.json","measuredAt":"2025-04-10","scope":"Registered model evaluation · EvalEval"},{"benchmark":"MMLU-Pro · mmlu pro","score":80.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/b5/69/b569e3ed-36d6-4b81-9343-2d241e116b18.json","measuredAt":"2026-06-30","scope":"Registered model evaluation · EvalEval"},{"benchmark":"LEXam · open question","score":47.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://lexam-benchmark.github.io/","measuredAt":"2026-06-02","scope":"Registered model evaluation · LEXam Leaderboard"},{"benchmark":"LEXam · mcq 4 choices","score":49.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://lexam-benchmark.github.io/","measuredAt":"2026-06-02","scope":"Registered model evaluation · LEXam Leaderboard"},{"benchmark":"SWE-bench_Pro · SWE Bench Pro","score":5.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://scale.com/leaderboard/swe_bench_pro_public","measuredAt":"2026-02-28","scope":"Registered model evaluation · SWE-Bench Pro official evaluation results"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"6f948657-6919-4b4d-bcd3-3a3a79839c6d","title":"Mark Zuckerberg — AI will write most Meta code in 18 months","href":"/intel/6f948657-6919-4b4d-bcd3-3a3a79839c6d","hook":"A frontier-lab CEO's candid take on Llama 4, benchmark gaming, open-source model economics, and how fast AI will take over code generation — useful signal fo…"},{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"dd3c95ad-0fac-49bd-9582-e0c78686e14f","title":"AutoAgent: Self-Optimizing AI Agents","href":"/intel/dd3c95ad-0fac-49bd-9582-e0c78686e14f","hook":"An open-source meta-agent that tunes task agents — prompts, tools, orchestration — until performance climbs, no human in the loop. A working reference if you…"}]}]},{"id":"muse","label":"Muse","models":[{"id":"meta/muse-spark-1.1-20260709","name":"Muse Spark 1.1","releaseLabel":"Spark 1.1","generationId":"1.1","generationLabel":"1.1","variantLabel":"Spark","createdAt":"2026-07-16T15:29:01.000Z","kind":"multimodal","contextLength":1048576,"inputModalities":["text","image","video","file","audio"],"outputModalities":["text"],"promptPerMillion":1.25,"completionPerMillion":4.25,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Muse Spark 1.1 is a multimodal reasoning model from Meta, built for agentic tasks. It accepts text, images, video, audio, and PDF documents and returns text, with a 1M-token context...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"meta/muse-spark-1.1","url":"https://openrouter.ai/meta/muse-spark-1.1","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":76,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1478,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 14 · 20,242 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"dd3c95ad-0fac-49bd-9582-e0c78686e14f","title":"AutoAgent: Self-Optimizing AI Agents","href":"/intel/dd3c95ad-0fac-49bd-9582-e0c78686e14f","hook":"An open-source meta-agent that tunes task agents — prompts, tools, orchestration — until performance climbs, no human in the loop. A working reference if you…"},{"id":"dcdafb0d-7527-43ee-a4a7-6d926aa5e691","title":"Meta-Harness: End-to-End Optimization of Model Harnesses","href":"/intel/dcdafb0d-7527-43ee-a4a7-6d926aa5e691","hook":"Harness engineering, automated: this optimizes the scaffolding around a model — what it stores, retrieves, and feeds back — rather than the weights. Read it …"},{"id":"5b24252c-82d9-48ef-bf7f-fc3593c9bb49","title":"The non-technical PM’s guide to building with Cursor | Zevi Arnovitz (Meta)","href":"/intel/5b24252c-82d9-48ef-bf7f-fc3593c9bb49","hook":"Breaks down a repeatable Cursor workflow — routing tasks across Claude and Gemini, automating prompts with slash commands, and having models peer-review each…"}]},{"id":"meta/muse-spark-1.2-20260805","name":"Muse Spark 1.2","releaseLabel":"Spark 1.2","generationId":"1.2","generationLabel":"1.2","variantLabel":"Spark","createdAt":"2026-08-05T19:48:07.000Z","kind":"multimodal","contextLength":1048576,"inputModalities":["text","image","video","file","audio"],"outputModalities":["text"],"promptPerMillion":1.25,"completionPerMillion":4.25,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Muse Spark 1.2 is a reasoning model from Meta, designed for complex agentic tasks. It accepts text, images, video, audio, and PDF documents, returns text, and offers a 1M-token context...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"meta/muse-spark-1.2","url":"https://openrouter.ai/meta/muse-spark-1.2","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":78.9,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"AA Intelligence Index","score":56.8,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/muse-spark-1-2","measuredAt":"2026-08-27","scope":"Current public index; highest tested effort configuration"},{"benchmark":"LMArena overall","score":1487,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 8 · 3,257 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"dd3c95ad-0fac-49bd-9582-e0c78686e14f","title":"AutoAgent: Self-Optimizing AI Agents","href":"/intel/dd3c95ad-0fac-49bd-9582-e0c78686e14f","hook":"An open-source meta-agent that tunes task agents — prompts, tools, orchestration — until performance climbs, no human in the loop. A working reference if you…"},{"id":"dcdafb0d-7527-43ee-a4a7-6d926aa5e691","title":"Meta-Harness: End-to-End Optimization of Model Harnesses","href":"/intel/dcdafb0d-7527-43ee-a4a7-6d926aa5e691","hook":"Harness engineering, automated: this optimizes the scaffolding around a model — what it stores, retrieves, and feeds back — rather than the weights. Read it …"},{"id":"5b24252c-82d9-48ef-bf7f-fc3593c9bb49","title":"The non-technical PM’s guide to building with Cursor | Zevi Arnovitz (Meta)","href":"/intel/5b24252c-82d9-48ef-bf7f-fc3593c9bb49","hook":"Breaks down a repeatable Cursor workflow — routing tasks across Claude and Gemini, automating prompts with slash commands, and having models peer-review each…"}]},{"id":"meta/muse-glimmer-30b-20260810","name":"Muse Glimmer 30B","releaseLabel":"Glimmer 30B","generationId":"30b","generationLabel":"30B","variantLabel":"Glimmer","createdAt":"2026-08-09T19:06:34.000Z","kind":"multimodal","contextLength":131072,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0.35,"completionPerMillion":1.5,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Muse Glimmer 30B is a dense, open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark and optimized for autonomous agents on consumer hardware. It is suited for long-horizon...","huggingFaceId":"meta-models/Muse-Glimmer-30B","routes":[{"provider":"OpenRouter","modelId":"meta/muse-glimmer-30b","url":"https://openrouter.ai/meta/muse-glimmer-30b","free":false,"batch":false}],"benchmarks":[{"benchmark":"terminal-bench-2.1 · terminalbench 2 1","score":51.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/meta-models/Muse-Glimmer-30B","measuredAt":"2026-08-11","scope":"Registered model evaluation · Model Card"},{"benchmark":"gpqa · diamond","score":83.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/meta-models/Muse-Glimmer-30B","measuredAt":"2026-08-10","scope":"Registered model evaluation · Muse Glimmer-30B model card"},{"benchmark":"MMMU_Pro · mmmu pro standard 10 options","score":74,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/meta-models/Muse-Glimmer-30B","measuredAt":"2026-08-10","scope":"Registered model evaluation · Muse Glimmer-30B model card"},{"benchmark":"aime_2026 · MathArena/aime 2026","score":94.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/meta-models/Muse-Glimmer-30B","measuredAt":"2026-08-10","scope":"Registered model evaluation · Muse Glimmer-30B model card"},{"benchmark":"SWE-bench_Verified · swe bench % resolved","score":76,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/meta-models/Muse-Glimmer-30B","measuredAt":"2026-08-10","scope":"Registered model evaluation · Muse Glimmer-30B model card"},{"benchmark":"SWE-bench_Pro · SWE Bench Pro","score":51.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/meta-models/Muse-Glimmer-30B","measuredAt":"2026-08-10","scope":"Registered model evaluation · Muse Glimmer-30B model card"},{"benchmark":"hle · hle","score":22,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/meta-models/Muse-Glimmer-30B","measuredAt":"2026-08-10","scope":"Registered model evaluation · Muse Glimmer-30B model card"},{"benchmark":"skillsbench · skillsbench v1 1","score":44.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/meta-models/Muse-Glimmer-30B","measuredAt":"2026-08-10","scope":"Registered model evaluation · Muse Glimmer-30B model card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"dd3c95ad-0fac-49bd-9582-e0c78686e14f","title":"AutoAgent: Self-Optimizing AI Agents","href":"/intel/dd3c95ad-0fac-49bd-9582-e0c78686e14f","hook":"An open-source meta-agent that tunes task agents — prompts, tools, orchestration — until performance climbs, no human in the loop. A working reference if you…"},{"id":"dcdafb0d-7527-43ee-a4a7-6d926aa5e691","title":"Meta-Harness: End-to-End Optimization of Model Harnesses","href":"/intel/dcdafb0d-7527-43ee-a4a7-6d926aa5e691","hook":"Harness engineering, automated: this optimizes the scaffolding around a model — what it stores, retrieves, and feeds back — rather than the weights. Read it …"},{"id":"5b24252c-82d9-48ef-bf7f-fc3593c9bb49","title":"The non-technical PM’s guide to building with Cursor | Zevi Arnovitz (Meta)","href":"/intel/5b24252c-82d9-48ef-bf7f-fc3593c9bb49","hook":"Breaks down a repeatable Cursor workflow — routing tasks across Claude and Gemini, automating prompts with slash commands, and having models peer-review each…"}]},{"id":"meta/muse-spark-1.2-contributor-20260805","name":"Muse Spark 1.2 Contributor","releaseLabel":"Spark 1.2 Contributor","generationId":"1.2","generationLabel":"1.2","variantLabel":"Contributor","createdAt":"2026-08-21T18:21:16.000Z","kind":"multimodal","contextLength":1048576,"inputModalities":["text","image","video","file","audio"],"outputModalities":["text"],"promptPerMillion":0.09999999999999999,"completionPerMillion":0.19999999999999998,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Muse Spark 1.2 contributor tier is a reasoning model from Meta designed for developers who want to start building at an even lower cost. It’s meaningfully cheaper than Muse Spark...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"meta/muse-spark-1.2-contributor","url":"https://openrouter.ai/meta/muse-spark-1.2-contributor","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"dd3c95ad-0fac-49bd-9582-e0c78686e14f","title":"AutoAgent: Self-Optimizing AI Agents","href":"/intel/dd3c95ad-0fac-49bd-9582-e0c78686e14f","hook":"An open-source meta-agent that tunes task agents — prompts, tools, orchestration — until performance climbs, no human in the loop. A working reference if you…"},{"id":"dcdafb0d-7527-43ee-a4a7-6d926aa5e691","title":"Meta-Harness: End-to-End Optimization of Model Harnesses","href":"/intel/dcdafb0d-7527-43ee-a4a7-6d926aa5e691","hook":"Harness engineering, automated: this optimizes the scaffolding around a model — what it stores, retrieves, and feeds back — rather than the weights. Read it …"},{"id":"5b24252c-82d9-48ef-bf7f-fc3593c9bb49","title":"The non-technical PM’s guide to building with Cursor | Zevi Arnovitz (Meta)","href":"/intel/5b24252c-82d9-48ef-bf7f-fc3593c9bb49","hook":"Breaks down a repeatable Cursor workflow — routing tasks across Claude and Gemini, automating prompts with slash commands, and having models peer-review each…"}]},{"id":"meta/muse-image-1.0-eval-20260824","name":"Muse Image","releaseLabel":"Image","generationId":"releases","generationLabel":"Releases","variantLabel":"Image","createdAt":"2026-08-26T17:15:32.000Z","kind":"image","contextLength":65536,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":2.39520958083832,"supportsTools":false,"supportsReasoning":false,"description":"Muse Image is an agentic image generation model from Meta that generates and edits images from text and reference images. Unlike single-pass image models, it reasons before it renders, breaking...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"meta/muse-image","url":"https://openrouter.ai/meta/muse-image","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[{"id":"dd3c95ad-0fac-49bd-9582-e0c78686e14f","title":"AutoAgent: Self-Optimizing AI Agents","href":"/intel/dd3c95ad-0fac-49bd-9582-e0c78686e14f","hook":"An open-source meta-agent that tunes task agents — prompts, tools, orchestration — until performance climbs, no human in the loop. A working reference if you…"},{"id":"dcdafb0d-7527-43ee-a4a7-6d926aa5e691","title":"Meta-Harness: End-to-End Optimization of Model Harnesses","href":"/intel/dcdafb0d-7527-43ee-a4a7-6d926aa5e691","hook":"Harness engineering, automated: this optimizes the scaffolding around a model — what it stores, retrieves, and feeds back — rather than the weights. Read it …"},{"id":"5b24252c-82d9-48ef-bf7f-fc3593c9bb49","title":"The non-technical PM’s guide to building with Cursor | Zevi Arnovitz (Meta)","href":"/intel/5b24252c-82d9-48ef-bf7f-fc3593c9bb49","hook":"Breaks down a repeatable Cursor workflow — routing tasks across Claude and Gemini, automating prompts with slash commands, and having models peer-review each…"}]}]},{"id":"llama-guard","label":"Llama Guard","models":[{"id":"meta/llama-guard-4-12b","name":"Llama Guard 4 12B","releaseLabel":"4 12B","generationId":"4","generationLabel":"4","variantLabel":"12B","createdAt":"2025-04-30T01:06:33.000Z","kind":"multimodal","contextLength":163840,"inputModalities":["image","text"],"outputModalities":["text"],"promptPerMillion":0.18,"completionPerMillion":0.18,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Llama Guard 4 is a Llama 4 Scout-derived multimodal pretrained model, fine-tuned for content safety classification. Similar to previous versions, it can be used to classify content in both LLM...","huggingFaceId":"meta-llama/Llama-Guard-4-12B","routes":[{"provider":"OpenRouter","modelId":"meta-llama/llama-guard-4-12b","url":"https://openrouter.ai/meta-llama/llama-guard-4-12b","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"dd3c95ad-0fac-49bd-9582-e0c78686e14f","title":"AutoAgent: Self-Optimizing AI Agents","href":"/intel/dd3c95ad-0fac-49bd-9582-e0c78686e14f","hook":"An open-source meta-agent that tunes task agents — prompts, tools, orchestration — until performance climbs, no human in the loop. A working reference if you…"},{"id":"dcdafb0d-7527-43ee-a4a7-6d926aa5e691","title":"Meta-Harness: End-to-End Optimization of Model Harnesses","href":"/intel/dcdafb0d-7527-43ee-a4a7-6d926aa5e691","hook":"Harness engineering, automated: this optimizes the scaffolding around a model — what it stores, retrieves, and feeds back — rather than the weights. Read it …"},{"id":"5b24252c-82d9-48ef-bf7f-fc3593c9bb49","title":"The non-technical PM’s guide to building with Cursor | Zevi Arnovitz (Meta)","href":"/intel/5b24252c-82d9-48ef-bf7f-fc3593c9bb49","hook":"Breaks down a repeatable Cursor workflow — routing tasks across Claude and Gemini, automating prompts with slash commands, and having models peer-review each…"}]}]}]},{"id":"mistralai","name":"Mistral AI","domain":"mistral.ai","major":true,"modelCount":25,"series":[{"id":"mistral","label":"Mistral","models":[{"id":"mistralai/mistral-large","name":"Mistral Large","releaseLabel":"Large","generationId":"releases","generationLabel":"Releases","variantLabel":"Large","createdAt":"2024-02-26T00:00:00.000Z","kind":"language","contextLength":128000,"inputModalities":["text","file"],"outputModalities":["text"],"promptPerMillion":2,"completionPerMillion":6,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"This is Mistral AI's flagship model, Mistral Large 2 (version `mistral-large-2407`). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"mistralai/mistral-large","url":"https://openrouter.ai/mistralai/mistral-large","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"mistralai/mistral-nemo","name":"Mistral Nemo","releaseLabel":"Nemo","generationId":"releases","generationLabel":"Releases","variantLabel":"Nemo","createdAt":"2024-07-19T00:00:00.000Z","kind":"language","contextLength":131072,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.019000000000000003,"completionPerMillion":0.03,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese,...","huggingFaceId":"mistralai/Mistral-Nemo-Instruct-2407","routes":[{"provider":"OpenRouter","modelId":"mistralai/mistral-nemo","url":"https://openrouter.ai/mistralai/mistral-nemo","free":false,"batch":false}],"benchmarks":[{"benchmark":"signalbench · src","score":0.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation · signalbench raw per-item responses"},{"benchmark":"signalbench · time","score":0.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation"},{"benchmark":"signalbench · access deny","score":0.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation"},{"benchmark":"signalbench · memory label","score":0.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation"},{"benchmark":"signalbench · injection","score":0.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation"},{"benchmark":"signalbench · bot policy","score":0.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/thamilvendhan/signalbench","measuredAt":"2026-07-08","scope":"Registered model evaluation"},{"benchmark":"MMLU-Pro · mmlu pro","score":44.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/0d/72/0d7207b9-dd9a-4a3e-bb04-0f5a94e886dc.json","measuredAt":"2026-06-30","scope":"Registered model evaluation · EvalEval"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"mistralai/mistral-large-2407","name":"Mistral Large 2407","releaseLabel":"Large 2407","generationId":"2407","generationLabel":"2407","variantLabel":"Large","createdAt":"2024-11-19T01:06:55.000Z","kind":"language","contextLength":131072,"inputModalities":["text","file"],"outputModalities":["text"],"promptPerMillion":2,"completionPerMillion":6,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"This is Mistral AI's flagship model, Mistral Large 2 (version mistral-large-2407). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"mistralai/mistral-large-2407","url":"https://openrouter.ai/mistralai/mistral-large-2407","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":49.9,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2024_08_31.csv","measuredAt":"2024-08-31","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1266,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 261 · 45,459 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"mistralai/mistral-small-24b-instruct-2501","name":"Mistral Small 3","releaseLabel":"Small 3","generationId":"24b","generationLabel":"24B","variantLabel":"Instruct 2501","createdAt":"2025-01-30T16:43:29.000Z","kind":"language","contextLength":32768,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.049999999999999996,"completionPerMillion":0.08,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Mistral Small 3 is a 24B-parameter language model optimized for low-latency performance across common AI tasks. Released under the Apache 2.0 license, it features both pre-trained and instruction-tuned versions designed...","huggingFaceId":"mistralai/Mistral-Small-24B-Instruct-2501","routes":[{"provider":"OpenRouter","modelId":"mistralai/mistral-small-24b-instruct-2501","url":"https://openrouter.ai/mistralai/mistral-small-24b-instruct-2501","free":false,"batch":false}],"benchmarks":[{"benchmark":"MMLU-Pro · mmlu pro","score":66.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/b7/50/b750fec7-95bc-4ec1-abb8-e580d1214beb.json","measuredAt":"2026-06-30","scope":"Registered model evaluation · EvalEval"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"mistralai/mistral-saba-2502","name":"Saba","releaseLabel":"Saba","generationId":"releases","generationLabel":"Releases","variantLabel":"Saba","createdAt":"2025-02-17T14:40:39.000Z","kind":"language","contextLength":32768,"inputModalities":["text","file"],"outputModalities":["text"],"promptPerMillion":0.19999999999999998,"completionPerMillion":0.6,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Mistral Saba is a 24B-parameter language model specifically designed for the Middle East and South Asia, delivering accurate and contextually relevant responses while maintaining efficient performance. Trained on curated regional...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"mistralai/mistral-saba","url":"https://openrouter.ai/mistralai/mistral-saba","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"mistralai/mistral-small-3.1-24b-instruct-2503","name":"Mistral Small 3.1 24B","releaseLabel":"Small 3.1 24B","generationId":"3.1","generationLabel":"3.1","variantLabel":"24B Instruct","createdAt":"2025-03-17T19:15:37.000Z","kind":"multimodal","contextLength":128000,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0.351,"completionPerMillion":0.5549999999999999,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Mistral Small 3.1 24B Instruct is an upgraded variant of Mistral Small 3 (2501), featuring 24 billion parameters with advanced multimodal capabilities. It provides state-of-the-art performance in text-based reasoning and...","huggingFaceId":"mistralai/Mistral-Small-3.1-24B-Instruct-2503","routes":[{"provider":"OpenRouter","modelId":"mistralai/mistral-small-3.1-24b-instruct","url":"https://openrouter.ai/mistralai/mistral-small-3.1-24b-instruct","free":false,"batch":false}],"benchmarks":[{"benchmark":"MMLU-Pro · mmlu pro","score":66.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/d1/e8/d1e83a09-a321-45ef-a93b-26f04d50e2d6.json","measuredAt":"2026-06-30","scope":"Registered model evaluation · EvalEval"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"mistralai/mistral-medium-3","name":"Mistral Medium 3","releaseLabel":"Medium 3","generationId":"3","generationLabel":"3","variantLabel":"Medium","createdAt":"2025-05-07T14:15:41.000Z","kind":"multimodal","contextLength":131072,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":0.39999999999999997,"completionPerMillion":2,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Mistral Medium 3 is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances state-of-the-art reasoning and multimodal performance with 8× lower cost...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"mistralai/mistral-medium-3","url":"https://openrouter.ai/mistralai/mistral-medium-3","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":50.2,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2025_05_30.csv","measuredAt":"2025-05-30","scope":"Mean across that public release's task columns"},{"benchmark":"AA Intelligence Index","score":30.4,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Current public index; highest tested effort configuration"},{"benchmark":"AA Agentic Index","score":19.2,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Current public agentic composite"},{"benchmark":"SciCode","score":39.6,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Scientific coding evaluation"},{"benchmark":"AA Omniscience Index","score":-36.8,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Current public knowledge and hallucination index"},{"benchmark":"MLCR overall","score":1.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Machine-learning code reasoning evaluation"},{"benchmark":"GDPval-AA normalized","score":21.8,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Normalized real-world knowledge-work evaluation"},{"benchmark":"τ²-bench","score":94.2,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Tool-agent benchmark across supported domains"},{"benchmark":"τ²-bench Banking","score":15.1,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Tool-agent benchmark in the banking domain"},{"benchmark":"Terminal-Bench Hard","score":33.3,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Hard terminal-agent tasks"},{"benchmark":"Terminal-Bench 2.1","score":50.6,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Terminal-agent task completion"},{"benchmark":"LCR","score":65.3,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Long-context reasoning evaluation"},{"benchmark":"IFBench","score":68.8,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Instruction-following evaluation"},{"benchmark":"Humanity's Last Exam","score":13.8,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Broad expert-level knowledge evaluation"},{"benchmark":"GPQA","score":74.8,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Graduate-level science questions"},{"benchmark":"CritPt","score":0,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Critical-point reasoning evaluation"},{"benchmark":"MMMU Pro","score":64.9,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Multimodal expert reasoning evaluation"},{"benchmark":"Analyst Agent","score":12.5,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Analyst workflow agent tasks"},{"benchmark":"Enterprise Ops Gym","score":33.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Enterprise operations agent tasks"},{"benchmark":"LMArena overall","score":1425,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 89 · 94,016 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"mistralai/mistral-small-3.2-24b-instruct-2506","name":"Mistral Small 3.2 24B","releaseLabel":"Small 3.2 24B","generationId":"3.2","generationLabel":"3.2","variantLabel":"24B Instruct","createdAt":"2025-06-20T18:10:16.000Z","kind":"multimodal","contextLength":131072,"inputModalities":["image","text"],"outputModalities":["text"],"promptPerMillion":0.075,"completionPerMillion":0.19999999999999998,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Mistral-Small-3.2-24B-Instruct-2506 is an updated 24B parameter model from Mistral optimized for instruction following, repetition reduction, and improved function calling. Compared to the 3.1 release, version 3.2 significantly improves accuracy on...","huggingFaceId":"mistralai/Mistral-Small-3.2-24B-Instruct-2506","routes":[{"provider":"OpenRouter","modelId":"mistralai/mistral-small-3.2-24b-instruct","url":"https://openrouter.ai/mistralai/mistral-small-3.2-24b-instruct","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"mistralai/mistral-medium-3.1","name":"Mistral Medium 3.1","releaseLabel":"Medium 3.1","generationId":"3.1","generationLabel":"3.1","variantLabel":"Medium","createdAt":"2025-08-13T14:33:59.000Z","kind":"multimodal","contextLength":131072,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":0.39999999999999997,"completionPerMillion":2,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Mistral Medium 3.1 is an updated version of Mistral Medium 3, which is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"mistralai/mistral-medium-3.1","url":"https://openrouter.ai/mistralai/mistral-medium-3.1","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":50.2,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2025_05_30.csv","measuredAt":"2025-05-30","scope":"Mean across that public release's task columns"},{"benchmark":"AA Intelligence Index","score":30.4,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Current public index; highest tested effort configuration"},{"benchmark":"AA Agentic Index","score":19.2,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Current public agentic composite"},{"benchmark":"SciCode","score":39.6,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Scientific coding evaluation"},{"benchmark":"AA Omniscience Index","score":-36.8,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Current public knowledge and hallucination index"},{"benchmark":"MLCR overall","score":1.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Machine-learning code reasoning evaluation"},{"benchmark":"GDPval-AA normalized","score":21.8,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Normalized real-world knowledge-work evaluation"},{"benchmark":"τ²-bench","score":94.2,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Tool-agent benchmark across supported domains"},{"benchmark":"τ²-bench Banking","score":15.1,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Tool-agent benchmark in the banking domain"},{"benchmark":"Terminal-Bench Hard","score":33.3,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Hard terminal-agent tasks"},{"benchmark":"Terminal-Bench 2.1","score":50.6,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Terminal-agent task completion"},{"benchmark":"LCR","score":65.3,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Long-context reasoning evaluation"},{"benchmark":"IFBench","score":68.8,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Instruction-following evaluation"},{"benchmark":"Humanity's Last Exam","score":13.8,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Broad expert-level knowledge evaluation"},{"benchmark":"GPQA","score":74.8,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Graduate-level science questions"},{"benchmark":"CritPt","score":0,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Critical-point reasoning evaluation"},{"benchmark":"MMMU Pro","score":64.9,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Multimodal expert reasoning evaluation"},{"benchmark":"Analyst Agent","score":12.5,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Analyst workflow agent tasks"},{"benchmark":"Enterprise Ops Gym","score":33.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Enterprise operations agent tasks"},{"benchmark":"LMArena overall","score":1425,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 89 · 94,016 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"mistralai/mistral-large-2512","name":"Mistral Large 3 2512","releaseLabel":"Large 3 2512","generationId":"2512","generationLabel":"2512","variantLabel":"Large 3","createdAt":"2025-12-01T21:27:52.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":0.5,"completionPerMillion":1.5,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Mistral Large 3 2512 is Mistral’s most capable model to date, featuring a sparse mixture-of-experts architecture with 41B active parameters (675B total), and released under the Apache 2.0 license.","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"mistralai/mistral-large-2512","url":"https://openrouter.ai/mistralai/mistral-large-2512","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"mistralai/mistral-small-2603","name":"Mistral Small 4","releaseLabel":"Small 4","generationId":"2603","generationLabel":"2603","variantLabel":"Small 4","createdAt":"2026-03-16T21:14:45.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0.15,"completionPerMillion":0.6,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Mistral Small 4 is the next major release in the Mistral Small family, unifying the capabilities of several flagship Mistral models into a single system. It combines strong reasoning from...","huggingFaceId":"mistralai/Mistral-Small-4-119B-2603","routes":[{"provider":"OpenRouter","modelId":"mistralai/mistral-small-2603","url":"https://openrouter.ai/mistralai/mistral-small-2603","free":false,"batch":false}],"benchmarks":[{"benchmark":"GPQA","score":71.2,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"gpqa · diamond","score":71.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/mistralai/Mistral-Small-4-119B-2603","measuredAt":"2026-03-16","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"mistralai/mistral-medium-3.5-20260430","name":"Mistral Medium 3.5","releaseLabel":"Medium 3.5","generationId":"3","generationLabel":"3","variantLabel":"5","createdAt":"2026-04-30T17:33:59.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":1.5,"completionPerMillion":7.5,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Mistral Medium 3.5 is a dense 128B instruction-following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"mistralai/mistral-medium-3-5","url":"https://openrouter.ai/mistralai/mistral-medium-3-5","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":50.2,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2025_05_30.csv","measuredAt":"2025-05-30","scope":"Mean across that public release's task columns"},{"benchmark":"AA Intelligence Index","score":30.4,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Current public index; highest tested effort configuration"},{"benchmark":"AA Agentic Index","score":19.2,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Current public agentic composite"},{"benchmark":"SciCode","score":39.6,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Scientific coding evaluation"},{"benchmark":"AA Omniscience Index","score":-36.8,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Current public knowledge and hallucination index"},{"benchmark":"MLCR overall","score":1.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Machine-learning code reasoning evaluation"},{"benchmark":"GDPval-AA normalized","score":21.8,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Normalized real-world knowledge-work evaluation"},{"benchmark":"τ²-bench","score":94.2,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Tool-agent benchmark across supported domains"},{"benchmark":"τ²-bench Banking","score":15.1,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Tool-agent benchmark in the banking domain"},{"benchmark":"Terminal-Bench Hard","score":33.3,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Hard terminal-agent tasks"},{"benchmark":"Terminal-Bench 2.1","score":50.6,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Terminal-agent task completion"},{"benchmark":"LCR","score":65.3,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Long-context reasoning evaluation"},{"benchmark":"IFBench","score":68.8,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Instruction-following evaluation"},{"benchmark":"Humanity's Last Exam","score":13.8,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Broad expert-level knowledge evaluation"},{"benchmark":"GPQA","score":74.8,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Graduate-level science questions"},{"benchmark":"CritPt","score":0,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Critical-point reasoning evaluation"},{"benchmark":"MMMU Pro","score":64.9,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Multimodal expert reasoning evaluation"},{"benchmark":"Analyst Agent","score":12.5,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Analyst workflow agent tasks"},{"benchmark":"Enterprise Ops Gym","score":33.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","measuredAt":"2026-08-27","scope":"Enterprise operations agent tasks"},{"benchmark":"LMArena overall","score":1425,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 89 · 94,016 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]},{"id":"ministral","label":"Ministral","models":[{"id":"mistralai/ministral-3b-2512","name":"Ministral 3 3B 2512","releaseLabel":"3 3B 2512","generationId":"3","generationLabel":"3","variantLabel":"3B","createdAt":"2025-12-02T13:19:20.000Z","kind":"multimodal","contextLength":131072,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0.09999999999999999,"completionPerMillion":0.09999999999999999,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"The smallest model in the Ministral 3 family, Ministral 3 3B is a powerful, efficient tiny language model with vision capabilities.","huggingFaceId":"mistralai/Ministral-3-3B-Instruct-2512","routes":[{"provider":"OpenRouter","modelId":"mistralai/ministral-3b-2512","url":"https://openrouter.ai/mistralai/ministral-3b-2512","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"mistralai/ministral-8b-2512","name":"Ministral 3 8B 2512","releaseLabel":"3 8B 2512","generationId":"3","generationLabel":"3","variantLabel":"8B","createdAt":"2025-12-02T13:20:54.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0.15,"completionPerMillion":0.15,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"A balanced model in the Ministral 3 family, Ministral 3 8B is a powerful, efficient tiny language model with vision capabilities.","huggingFaceId":"mistralai/Ministral-3-8B-Instruct-2512","routes":[{"provider":"OpenRouter","modelId":"mistralai/ministral-8b-2512","url":"https://openrouter.ai/mistralai/ministral-8b-2512","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"mistralai/ministral-14b-2512","name":"Ministral 3 14B 2512","releaseLabel":"3 14B 2512","generationId":"3","generationLabel":"3","variantLabel":"14B","createdAt":"2025-12-02T13:22:15.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0.19999999999999998,"completionPerMillion":0.19999999999999998,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"The largest model in the Ministral 3 family, Ministral 3 14B offers frontier capabilities and performance comparable to its larger Mistral Small 3.2 24B counterpart. A powerful and efficient language...","huggingFaceId":"mistralai/Ministral-3-14B-Instruct-2512","routes":[{"provider":"OpenRouter","modelId":"mistralai/ministral-14b-2512","url":"https://openrouter.ai/mistralai/ministral-14b-2512","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]},{"id":"voxtral-transcribe","label":"Voxtral Transcribe","models":[{"id":"mistralai/voxtral-mini-transcribe-2602","name":"Voxtral Mini Transcribe","releaseLabel":"Voxtral Mini Transcribe","generationId":"releases","generationLabel":"Releases","variantLabel":"Voxtral Mini Transcribe","createdAt":"2026-05-15T20:30:24.000Z","kind":"audio","contextLength":0,"inputModalities":["audio"],"outputModalities":["transcription"],"promptPerMillion":3000,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Voxtral Mini Transcribe is Mistral's speech-to-text model, derived from the Voxtral Mini family. It accepts audio input and returns transcribed text via the standard transcription API. Suited for transcribing meetings,...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"mistralai/voxtral-mini-transcribe","url":"https://openrouter.ai/mistralai/voxtral-mini-transcribe","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[]},{"id":"mistralai/voxtral-mini-3b-2507-20260813","name":"Voxtral Mini 3B 2507","releaseLabel":"Voxtral Mini 3B 2507","generationId":"2507","generationLabel":"2507","variantLabel":"Mini 3B","createdAt":"2026-08-13T20:46:20.000Z","kind":"audio","contextLength":0,"inputModalities":["audio"],"outputModalities":["transcription"],"promptPerMillion":16.666700000000002,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Voxtral Mini 3B 2507 is a speech and audio understanding model from Mistral AI. It is suited for transcription, translation, and compact audio processing workloads.","huggingFaceId":"mistralai/Voxtral-Mini-3B-2507","routes":[{"provider":"OpenRouter","modelId":"mistralai/voxtral-mini-3b-2507","url":"https://openrouter.ai/mistralai/voxtral-mini-3b-2507","free":false,"batch":false}],"benchmarks":[{"benchmark":"open-asr-leaderboard · mean wer","score":7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2025-07-01","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · ami wer","score":16.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2025-07-01","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · earnings22 wer","score":10.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2025-07-01","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · gigaspeech wer","score":10.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2025-07-01","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · librispeech clean wer","score":1.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2025-07-01","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · librispeech other wer","score":4.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2025-07-01","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · spgispeech wer","score":2.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2025-07-01","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · tedlium wer","score":3.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2025-07-01","scope":"Registered model evaluation · open-asr-leaderboard"}],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[]},{"id":"mistralai/voxtral-small-24b-2507-stt-20260813","name":"Voxtral Small 24B 2507 STT","releaseLabel":"Voxtral Small 24B 2507 STT","generationId":"2507","generationLabel":"2507","variantLabel":"Small 24B","createdAt":"2026-08-13T20:46:42.000Z","kind":"audio","contextLength":0,"inputModalities":["audio"],"outputModalities":["transcription"],"promptPerMillion":50,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Voxtral Small 24B 2507 STT is a speech transcription model from Mistral AI. It is suited for transcription, translation, and audio understanding workloads that benefit from its larger model capacity.","huggingFaceId":"mistralai/Voxtral-Small-24B-2507","routes":[{"provider":"OpenRouter","modelId":"mistralai/voxtral-small-24b-2507-stt","url":"https://openrouter.ai/mistralai/voxtral-small-24b-2507-stt","free":false,"batch":false}],"benchmarks":[{"benchmark":"open-asr-leaderboard · mean wer","score":6.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2025-07-01","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · rtfx","score":54.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2025-07-01","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · ami wer","score":15.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2025-07-01","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · earnings22 wer","score":10.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2025-07-01","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · gigaspeech wer","score":9.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2025-07-01","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · librispeech clean wer","score":1.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2025-07-01","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · librispeech other wer","score":3.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2025-07-01","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · spgispeech wer","score":2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2025-07-01","scope":"Registered model evaluation · open-asr-leaderboard"}],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[]}]},{"id":"codestral","label":"Codestral","models":[{"id":"mistralai/codestral-2508","name":"Codestral 2508","releaseLabel":"2508","generationId":"2508","generationLabel":"2508","variantLabel":"Base","createdAt":"2025-08-01T20:20:30.000Z","kind":"language","contextLength":256000,"inputModalities":["text","file"],"outputModalities":["text"],"promptPerMillion":0.3,"completionPerMillion":0.8999999999999999,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Mistral's cutting-edge language model for coding released end of July 2025. Codestral specializes in low-latency, high-frequency tasks such as fill-in-the-middle (FIM), code correction and test generation. [Blog Post](https://mistral.ai/news/codestral-25-08)","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"mistralai/codestral-2508","url":"https://openrouter.ai/mistralai/codestral-2508","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]},{"id":"codestral-embed","label":"Codestral Embed","models":[{"id":"mistralai/codestral-embed-2505","name":"Codestral Embed 2505","releaseLabel":"2505","generationId":"2505","generationLabel":"2505","variantLabel":"Base","createdAt":"2025-10-30T22:47:40.000Z","kind":"embedding","contextLength":8192,"inputModalities":["text"],"outputModalities":["embeddings"],"promptPerMillion":0.15,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Mistral Codestral Embed is specially designed for code, perfect for embedding code databases, repositories, and powering coding assistants with state-of-the-art retrieval.","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"mistralai/codestral-embed-2505","url":"https://openrouter.ai/mistralai/codestral-embed-2505","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[]}]},{"id":"devstral","label":"Devstral","models":[{"id":"mistralai/devstral-2512","name":"Devstral 2 2512","releaseLabel":"2 2512","generationId":"2512","generationLabel":"2512","variantLabel":"2","createdAt":"2025-12-09T13:03:39.000Z","kind":"language","contextLength":262144,"inputModalities":["text","file"],"outputModalities":["text"],"promptPerMillion":0.44,"completionPerMillion":2.2,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Devstral 2 is a state-of-the-art open-source model by Mistral AI specializing in agentic coding. It is a 123B-parameter dense transformer model supporting a 256K context window. Devstral 2 supports exploring...","huggingFaceId":"mistralai/Devstral-2-123B-Instruct-2512","routes":[{"provider":"OpenRouter","modelId":"mistralai/devstral-2512","url":"https://openrouter.ai/mistralai/devstral-2512","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]},{"id":"mistral-embed","label":"Mistral Embed","models":[{"id":"mistralai/mistral-embed-2312","name":"Mistral Embed 2312","releaseLabel":"2312","generationId":"2312","generationLabel":"2312","variantLabel":"Base","createdAt":"2025-10-31T21:03:42.000Z","kind":"embedding","contextLength":8192,"inputModalities":["text"],"outputModalities":["embeddings"],"promptPerMillion":0.09999999999999999,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Mistral Embed is a specialized embedding model for text data, optimized for semantic search and RAG applications. Developed by Mistral AI in late 2023, it produces 1024-dimensional vectors that effectively...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"mistralai/mistral-embed-2312","url":"https://openrouter.ai/mistralai/mistral-embed-2312","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[]}]},{"id":"mixtral","label":"Mixtral","models":[{"id":"mistralai/mixtral-8x22b-instruct","name":"Mixtral 8x22B Instruct","releaseLabel":"8x22B Instruct","generationId":"initial","generationLabel":"Initial","variantLabel":"8x22b Instruct","createdAt":"2024-04-17T00:00:00.000Z","kind":"language","contextLength":65536,"inputModalities":["text","file"],"outputModalities":["text"],"promptPerMillion":2,"completionPerMillion":6,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Mistral's official instruct fine-tuned version of [Mixtral 8x22B](/models/mistralai/mixtral-8x22b). It uses 39B active parameters out of 141B, offering unparalleled cost efficiency for its size. Its strengths include: - strong math, coding,...","huggingFaceId":"mistralai/Mixtral-8x22B-Instruct-v0.1","routes":[{"provider":"OpenRouter","modelId":"mistralai/mixtral-8x22b-instruct","url":"https://openrouter.ai/mistralai/mixtral-8x22b-instruct","free":false,"batch":false}],"benchmarks":[{"benchmark":"MMLU-Pro · mmlu pro","score":56.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/8d/eb/8deb1927-af46-4cfc-a733-999f60ed4964.json","measuredAt":"2026-06-30","scope":"Registered model evaluation · EvalEval"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]},{"id":"voxtral","label":"Voxtral","models":[{"id":"mistralai/voxtral-small-24b-2507","name":"Voxtral Small 24B 2507","releaseLabel":"Small 24B 2507","generationId":"24b","generationLabel":"24B","variantLabel":"2507","createdAt":"2025-10-30T14:39:04.000Z","kind":"multimodal","contextLength":32000,"inputModalities":["text","audio","file"],"outputModalities":["text"],"promptPerMillion":0.09999999999999999,"completionPerMillion":0.3,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Voxtral Small is an enhancement of Mistral Small 3, incorporating state-of-the-art audio input capabilities while retaining best-in-class text performance. It excels at speech transcription, translation and audio understanding. Input audio...","huggingFaceId":"mistralai/Voxtral-Small-24B-2507","routes":[{"provider":"OpenRouter","modelId":"mistralai/voxtral-small-24b-2507","url":"https://openrouter.ai/mistralai/voxtral-small-24b-2507","free":false,"batch":false}],"benchmarks":[{"benchmark":"open-asr-leaderboard · mean wer","score":6.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2025-07-01","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · rtfx","score":54.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2025-07-01","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · ami wer","score":15.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2025-07-01","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · earnings22 wer","score":10.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2025-07-01","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · gigaspeech wer","score":9.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2025-07-01","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · librispeech clean wer","score":1.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2025-07-01","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · librispeech other wer","score":3.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2025-07-01","scope":"Registered model evaluation · open-asr-leaderboard"},{"benchmark":"open-asr-leaderboard · spgispeech wer","score":2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/hf-audio","measuredAt":"2025-07-01","scope":"Registered model evaluation · open-asr-leaderboard"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]},{"id":"voxtral-tts","label":"Voxtral TTS","models":[{"id":"mistralai/voxtral-mini-tts-2603","name":"Voxtral Mini TTS","releaseLabel":"Voxtral Mini TTS","generationId":"2603","generationLabel":"2603","variantLabel":"Mini","createdAt":"2026-04-19T04:02:17.000Z","kind":"audio","contextLength":4096,"inputModalities":["text"],"outputModalities":["speech"],"promptPerMillion":16,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Voxtral Mini TTS is Mistral's text-to-speech model featuring zero-shot voice cloning and multilingual support. It converts text input into natural-sounding audio output.","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"mistralai/voxtral-mini-tts-2603","url":"https://openrouter.ai/mistralai/voxtral-mini-tts-2603","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[]}]}]},{"id":"z-ai","name":"Z.ai","domain":"z.ai","major":true,"modelCount":14,"series":[{"id":"glm","label":"GLM","models":[{"id":"z-ai/glm-4.5-air","name":"GLM 4.5 Air","releaseLabel":"4.5 Air","generationId":"4.5","generationLabel":"4.5","variantLabel":"Air","createdAt":"2025-07-25T19:20:58.000Z","kind":"language","contextLength":131072,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.13,"completionPerMillion":0.85,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...","huggingFaceId":"zai-org/GLM-4.5-Air","routes":[{"provider":"OpenRouter","modelId":"z-ai/glm-4.5-air","url":"https://openrouter.ai/z-ai/glm-4.5-air","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":61.2,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2025_05_30.csv","measuredAt":"2025-05-30","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1383,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 151 · 31,190 votes"},{"benchmark":"gpqa · diamond","score":71.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/9c/77/9c7740fd-9cad-445c-9681-eb576e2a110e.json","measuredAt":"2026-04-19","scope":"Registered model evaluation · EvalEval"},{"benchmark":"hle · none","score":8.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/e9/03/e90308ab-e898-4434-aa93-13d131887737.json","measuredAt":"2025-08-13","scope":"Registered model evaluation · EvalEval"},{"benchmark":"MMLU-Pro · mmlu pro","score":81.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/07/43/074397da-f5e5-441c-bc45-ac16b599ca7c.json","measuredAt":"2026-06-30","scope":"Registered model evaluation · EvalEval"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"a45ca9db-1953-484a-9426-cab823615321","title":"GLM-5.2","href":"/intel/a45ca9db-1953-484a-9426-cab823615321","hook":"Where open-weight models actually stand against the closed frontier on agentic coding, argued from GLM-5.2's benchmark results."}]},{"id":"z-ai/glm-4.5","name":"GLM 4.5","releaseLabel":"4.5","generationId":"4.5","generationLabel":"4.5","variantLabel":"Base","createdAt":"2025-07-25T19:22:27.000Z","kind":"language","contextLength":131072,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.6,"completionPerMillion":2.2,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GLM-4.5 is our latest flagship foundation model, purpose-built for agent-based applications. It leverages a Mixture-of-Experts (MoE) architecture and supports a context length of up to 128k tokens. GLM-4.5 delivers significantly...","huggingFaceId":"zai-org/GLM-4.5","routes":[{"provider":"OpenRouter","modelId":"z-ai/glm-4.5","url":"https://openrouter.ai/z-ai/glm-4.5","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":65.9,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2025_05_30.csv","measuredAt":"2025-05-30","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1429,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 80 · 24,403 votes"},{"benchmark":"hle · none","score":8.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/93/cb/93cbdfdf-9ec6-4afb-87f4-d771cf29a378.json","measuredAt":"2025-08-13","scope":"Registered model evaluation · EvalEval"},{"benchmark":"MMLU-Pro · mmlu pro","score":84.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/04/c7/04c77f44-dff1-4cc7-80a1-35f4e9e9237f.json","measuredAt":"2026-06-30","scope":"Registered model evaluation · EvalEval"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"a45ca9db-1953-484a-9426-cab823615321","title":"GLM-5.2","href":"/intel/a45ca9db-1953-484a-9426-cab823615321","hook":"Where open-weight models actually stand against the closed frontier on agentic coding, argued from GLM-5.2's benchmark results."}]},{"id":"z-ai/glm-4.5v","name":"GLM 4.5V","releaseLabel":"4.5V","generationId":"4.5v","generationLabel":"4.5v","variantLabel":"Base","createdAt":"2025-08-11T14:24:48.000Z","kind":"multimodal","contextLength":65536,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0.6,"completionPerMillion":1.7999999999999998,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GLM-4.5V is a vision-language foundation model for multimodal agent applications. Built on a Mixture-of-Experts (MoE) architecture with 106B parameters and 12B activated parameters, it achieves state-of-the-art results in video understanding,...","huggingFaceId":"zai-org/GLM-4.5V","routes":[{"provider":"OpenRouter","modelId":"z-ai/glm-4.5v","url":"https://openrouter.ai/z-ai/glm-4.5v","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1334,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 202 · 4,968 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"a45ca9db-1953-484a-9426-cab823615321","title":"GLM-5.2","href":"/intel/a45ca9db-1953-484a-9426-cab823615321","hook":"Where open-weight models actually stand against the closed frontier on agentic coding, argued from GLM-5.2's benchmark results."}]},{"id":"z-ai/glm-4.6","name":"GLM 4.6","releaseLabel":"4.6","generationId":"4.6","generationLabel":"4.6","variantLabel":"Base","createdAt":"2025-09-30T12:32:56.000Z","kind":"language","contextLength":204800,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.43,"completionPerMillion":1.75,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...","huggingFaceId":"zai-org/GLM-4.6","routes":[{"provider":"OpenRouter","modelId":"z-ai/glm-4.6","url":"https://openrouter.ai/z-ai/glm-4.6","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":54.7,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1440,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 62 · 35,830 votes"},{"benchmark":"SWE-bench Pro","score":9.7,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"Terminal-Bench","score":24.5,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"apex-agents · apex-agents","score":4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://www.mercor.com/apex/apex-agents-leaderboard/","measuredAt":"2026-04-22","scope":"Registered model evaluation · APEX-Agents Leaderboard"},{"benchmark":"SWE-bench_Pro · SWE Bench Pro","score":9.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://scale.com/leaderboard/swe_bench_pro_public","measuredAt":"2026-02-28","scope":"Registered model evaluation · SWE-Bench Pro official evaluation results"},{"benchmark":"terminal-bench-2.0 · terminalbench 2","score":24.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://www.tbench.ai/leaderboard/terminal-bench/2.0","measuredAt":"2025-11-01","scope":"Registered model evaluation · Terminal-Bench Leaderboard"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"a45ca9db-1953-484a-9426-cab823615321","title":"GLM-5.2","href":"/intel/a45ca9db-1953-484a-9426-cab823615321","hook":"Where open-weight models actually stand against the closed frontier on agentic coding, argued from GLM-5.2's benchmark results."}]},{"id":"z-ai/glm-4.6-20251208","name":"GLM 4.6V","releaseLabel":"4.6V","generationId":"4.6v","generationLabel":"4.6v","variantLabel":"Base","createdAt":"2025-12-08T15:24:22.000Z","kind":"multimodal","contextLength":131072,"inputModalities":["image","text","video"],"outputModalities":["text"],"promptPerMillion":0.3,"completionPerMillion":0.8999999999999999,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GLM-4.6V is a large multimodal model designed for high-fidelity visual understanding and long-context reasoning across images, documents, and mixed media. It supports up to 128K tokens, processes complex page layouts...","huggingFaceId":"zai-org/GLM-4.6V","routes":[{"provider":"OpenRouter","modelId":"z-ai/glm-4.6v","url":"https://openrouter.ai/z-ai/glm-4.6v","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":38.9,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1375,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 160 · 2,817 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"a45ca9db-1953-484a-9426-cab823615321","title":"GLM-5.2","href":"/intel/a45ca9db-1953-484a-9426-cab823615321","hook":"Where open-weight models actually stand against the closed frontier on agentic coding, argued from GLM-5.2's benchmark results."}]},{"id":"z-ai/glm-4.7-20251222","name":"GLM 4.7","releaseLabel":"4.7","generationId":"4.7","generationLabel":"4.7","variantLabel":"Base","createdAt":"2025-12-22T04:33:34.000Z","kind":"language","contextLength":204800,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.39999999999999997,"completionPerMillion":1.75,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...","huggingFaceId":"zai-org/GLM-4.7","routes":[{"provider":"OpenRouter","modelId":"z-ai/glm-4.7","url":"https://openrouter.ai/z-ai/glm-4.7","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":57.3,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1435,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 75 · 12,155 votes"},{"benchmark":"EvasionBench","score":82.9,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"GPQA","score":85.7,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"Humanity's Last Exam","score":24.8,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"MMLU-Pro","score":84.3,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"SWE-bench Verified","score":73.8,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"Terminal-Bench","score":33.4,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"gpqa · diamond","score":85.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/zai-org/GLM-4.7","measuredAt":"2026-01-21","scope":"Registered model evaluation · Model Card"},{"benchmark":"hle · hle","score":24.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/zai-org/GLM-4.7","measuredAt":"2026-01-14","scope":"Registered model evaluation · Model Card"},{"benchmark":"apex-agents · apex-agents","score":3.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://www.mercor.com/apex/apex-agents-leaderboard/","measuredAt":"2026-04-22","scope":"Registered model evaluation · APEX-Agents Leaderboard"},{"benchmark":"APEX-v1-extended · apex-v1","score":51.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://www.mercor.com/apex/apex-v1-leaderboard/","measuredAt":"2026-04-22","scope":"Registered model evaluation · APEX Leaderboard"},{"benchmark":"SWE-bench_Verified · swe bench % resolved","score":73.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/zai-org/GLM-4.7","measuredAt":"2026-03-17","scope":"Registered model evaluation · Model Card"},{"benchmark":"terminal-bench-2.0 · terminalbench 2","score":33.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://www.tbench.ai/leaderboard/terminal-bench/2.0","measuredAt":"2026-01-28","scope":"Registered model evaluation · Terminal-Bench Leaderboard"},{"benchmark":"EvasionBench · evasion bench","score":82.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://arxiv.org/abs/2601.09142","measuredAt":"2026-02-10","scope":"Registered model evaluation · EvasionBench Paper"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"a45ca9db-1953-484a-9426-cab823615321","title":"GLM-5.2","href":"/intel/a45ca9db-1953-484a-9426-cab823615321","hook":"Where open-weight models actually stand against the closed frontier on agentic coding, argued from GLM-5.2's benchmark results."}]},{"id":"z-ai/glm-4.7-flash-20260119","name":"GLM 4.7 Flash","releaseLabel":"4.7 Flash","generationId":"4.7","generationLabel":"4.7","variantLabel":"Flash","createdAt":"2026-01-19T14:45:13.000Z","kind":"language","contextLength":202752,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.06,"completionPerMillion":0.39999999999999997,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...","huggingFaceId":"zai-org/GLM-4.7-Flash","routes":[{"provider":"OpenRouter","modelId":"z-ai/glm-4.7-flash","url":"https://openrouter.ai/z-ai/glm-4.7-flash","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1353,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 184 · 11,747 votes"},{"benchmark":"GPQA","score":75.2,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"Humanity's Last Exam","score":14.4,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"SWE-bench Verified","score":59.2,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"gpqa · diamond","score":75.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/zai-org/GLM-4.7-Flash","measuredAt":"2026-01-27","scope":"Registered model evaluation · Model Card"},{"benchmark":"hle · hle","score":14.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/zai-org/GLM-4.7-Flash","measuredAt":"2026-01-28","scope":"Registered model evaluation · Model Card"},{"benchmark":"SWE-bench_Verified · swe bench % resolved","score":59.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/zai-org/GLM-4.7-Flash","measuredAt":"2026-03-18","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"a45ca9db-1953-484a-9426-cab823615321","title":"GLM-5.2","href":"/intel/a45ca9db-1953-484a-9426-cab823615321","hook":"Where open-weight models actually stand against the closed frontier on agentic coding, argued from GLM-5.2's benchmark results."}]},{"id":"z-ai/glm-5-20260211","name":"GLM 5","releaseLabel":"5","generationId":"5","generationLabel":"5","variantLabel":"Base","createdAt":"2026-02-11T16:59:42.000Z","kind":"language","contextLength":204800,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.6,"completionPerMillion":1.92,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...","huggingFaceId":"zai-org/GLM-5","routes":[{"provider":"OpenRouter","modelId":"z-ai/glm-5","url":"https://openrouter.ai/z-ai/glm-5","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":68.7,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1445,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 48 · 27,850 votes"},{"benchmark":"AIME 2026","score":95.8,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"GPQA","score":86,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"Humanity's Last Exam","score":30.5,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"HMMT 2026","score":86.4,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"SWE-bench Verified","score":72.8,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"Terminal-Bench","score":52.4,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"aime_2026 · MathArena/aime 2026","score":95.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://matharena.ai/?comp=aime--aime_2026","measuredAt":"2026-02-18","scope":"Registered model evaluation · Official MathArena Evaluation"},{"benchmark":"hmmt_feb_2026 · MathArena/hmmt feb 2026","score":86.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://matharena.ai/?comp=hmmt--hmmt_feb_2026","measuredAt":"2026-02-23","scope":"Registered model evaluation · Official MathArena Evaluation"},{"benchmark":"gpqa · diamond","score":86,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/zai-org/GLM-5","measuredAt":"2026-02-13","scope":"Registered model evaluation · Model Card"},{"benchmark":"hle · hle","score":30.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/zai-org/GLM-5","measuredAt":"2026-02-13","scope":"Registered model evaluation · Model Card"},{"benchmark":"SWE-bench_Multilingual · swe bench multilingual % resolved","score":73.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/zai-org/GLM-5","measuredAt":"2026-08-11","scope":"Registered model evaluation · Model Card"},{"benchmark":"SWE-bench_Verified · swe bench % resolved","score":72.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://www.swebench.com/","measuredAt":"2026-03-24","scope":"Registered model evaluation · SWE-Bench official evaluation"},{"benchmark":"terminal-bench-2.0 · terminal bench","score":52.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://www.tbench.ai/leaderboard/terminal-bench/2.0","measuredAt":"2026-02-23","scope":"Registered model evaluation · Terminal-Bench Leaderboard"},{"benchmark":"terminal-bench-2.0 · terminalbench 2","score":52.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://www.tbench.ai/leaderboard/terminal-bench/2.0","measuredAt":"2026-02-23","scope":"Registered model evaluation · Terminal-Bench Leaderboard"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"a45ca9db-1953-484a-9426-cab823615321","title":"GLM-5.2","href":"/intel/a45ca9db-1953-484a-9426-cab823615321","hook":"Where open-weight models actually stand against the closed frontier on agentic coding, argued from GLM-5.2's benchmark results."}]},{"id":"z-ai/glm-5-turbo-20260315","name":"GLM 5 Turbo","releaseLabel":"5 Turbo","generationId":"5","generationLabel":"5","variantLabel":"Turbo","createdAt":"2026-03-15T14:06:13.000Z","kind":"language","contextLength":202752,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":1.2,"completionPerMillion":4,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GLM-5 Turbo is a new model from Z.ai designed for fast inference and strong performance in agent-driven environments such as OpenClaw scenarios. It is deeply optimized for real-world agent workflows...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"z-ai/glm-5-turbo","url":"https://openrouter.ai/z-ai/glm-5-turbo","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"a45ca9db-1953-484a-9426-cab823615321","title":"GLM-5.2","href":"/intel/a45ca9db-1953-484a-9426-cab823615321","hook":"Where open-weight models actually stand against the closed frontier on agentic coding, argued from GLM-5.2's benchmark results."}]},{"id":"z-ai/glm-5v-turbo-20260401","name":"GLM 5V Turbo","releaseLabel":"5V Turbo","generationId":"5v","generationLabel":"5v","variantLabel":"Turbo","createdAt":"2026-04-01T16:37:38.000Z","kind":"multimodal","contextLength":202752,"inputModalities":["image","text","video"],"outputModalities":["text"],"promptPerMillion":1.2,"completionPerMillion":4,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GLM-5V-Turbo is Z.ai’s first native multimodal agent foundation model, built for vision-based coding and agent-driven tasks. It natively handles image, video, and text inputs, excels at long-horizon planning, complex coding,...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"z-ai/glm-5v-turbo","url":"https://openrouter.ai/z-ai/glm-5v-turbo","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":48.8,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1437,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 71 · 9,380 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"a45ca9db-1953-484a-9426-cab823615321","title":"GLM-5.2","href":"/intel/a45ca9db-1953-484a-9426-cab823615321","hook":"Where open-weight models actually stand against the closed frontier on agentic coding, argued from GLM-5.2's benchmark results."}]},{"id":"z-ai/glm-5.1-20260406","name":"GLM 5.1","releaseLabel":"5.1","generationId":"5.1","generationLabel":"5.1","variantLabel":"Base","createdAt":"2026-04-07T16:07:05.000Z","kind":"language","contextLength":204800,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":1.26,"completionPerMillion":3.9600000000000004,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...","huggingFaceId":"zai-org/GLM-5.1","routes":[{"provider":"OpenRouter","modelId":"z-ai/glm-5.1","url":"https://openrouter.ai/z-ai/glm-5.1","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":70.6,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1464,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 29 · 42,267 votes"},{"benchmark":"aime_2026 · MathArena/aime 2026","score":95.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/zai-org/GLM-5.1","measuredAt":"2026-04-07","scope":"Registered model evaluation · Model Card"},{"benchmark":"hmmt_feb_2026 · MathArena/hmmt feb 2026","score":82.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/zai-org/GLM-5.1","measuredAt":"2026-04-07","scope":"Registered model evaluation · Model Card"},{"benchmark":"Claw-Eval · general","score":62.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://claw-eval.github.io","measuredAt":"2026-04-23","scope":"Registered model evaluation · Claw-Eval Leaderboard"},{"benchmark":"Claw-Eval · multi turn","score":60.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://claw-eval.github.io","measuredAt":"2026-04-23","scope":"Registered model evaluation · Claw-Eval Leaderboard"},{"benchmark":"gpqa · diamond","score":86.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/zai-org/GLM-5.1","measuredAt":"2026-04-07","scope":"Registered model evaluation · Model Card"},{"benchmark":"hle · hle","score":31,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/zai-org/GLM-5.1","measuredAt":"2026-04-07","scope":"Registered model evaluation · Model Card"},{"benchmark":"SWE-bench_Pro · SWE Bench Pro","score":58.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/zai-org/GLM-5.1","measuredAt":"2026-04-08","scope":"Registered model evaluation · Model Card"},{"benchmark":"terminal-bench-2.0 · terminalbench 2","score":63.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/zai-org/GLM-5.1","measuredAt":"2026-04-07","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"a45ca9db-1953-484a-9426-cab823615321","title":"GLM-5.2","href":"/intel/a45ca9db-1953-484a-9426-cab823615321","hook":"Where open-weight models actually stand against the closed frontier on agentic coding, argued from GLM-5.2's benchmark results."}]},{"id":"z-ai/glm-5.2-20260616","name":"GLM 5.2","releaseLabel":"5.2","generationId":"5.2","generationLabel":"5.2","variantLabel":"Base","createdAt":"2026-06-16T17:45:30.000Z","kind":"language","contextLength":1048576,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":1.19,"completionPerMillion":3.74,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","huggingFaceId":"zai-org/GLM-5.2","routes":[{"provider":"OpenRouter","modelId":"z-ai/glm-5.2","url":"https://openrouter.ai/z-ai/glm-5.2","free":false,"batch":false},{"provider":"OpenRouter","modelId":"z-ai/glm-5.2:free","url":"https://openrouter.ai/z-ai/glm-5.2:free","free":true,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":73.4,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"deep-swe · deep swe","score":46.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/zai-org/GLM-5.2","measuredAt":"2026-06-22","scope":"Registered model evaluation · Model Card"},{"benchmark":"gpqa · diamond","score":91.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/zai-org/GLM-5.2","measuredAt":"2026-06-22","scope":"Registered model evaluation · Model Card"},{"benchmark":"hle · hle","score":40.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/zai-org/GLM-5.2","measuredAt":"2026-06-22","scope":"Registered model evaluation · Model Card"},{"benchmark":"SWE-bench_Pro · SWE Bench Pro","score":62.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/zai-org/GLM-5.2","measuredAt":"2026-06-22","scope":"Registered model evaluation · Model Card"},{"benchmark":"RedlineBench · redline overall","score":45.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/zai-org/GLM-5.2","measuredAt":"2026-06-19","scope":"Registered model evaluation · Model Card"},{"benchmark":"terminal-bench-2.1 · terminalbench 2 1","score":81,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/zai-org/GLM-5.2","measuredAt":"2026-07-02","scope":"Registered model evaluation · Model Card"},{"benchmark":"WildClawBench · overall","score":54.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://internlm.github.io/WildClawBench","measuredAt":"2026-08-11","scope":"Registered model evaluation · WildClawBench"},{"benchmark":"Long-Horizon-Terminal-Bench · lhtb solved","score":1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://zli12321.github.io/LHTB/leaderboard.html","measuredAt":"2026-07-16","scope":"Registered model evaluation · LHTB leaderboard"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"a45ca9db-1953-484a-9426-cab823615321","title":"GLM-5.2","href":"/intel/a45ca9db-1953-484a-9426-cab823615321","hook":"Where open-weight models actually stand against the closed frontier on agentic coding, argued from GLM-5.2's benchmark results."}]},{"id":"z-ai/glm-5.3-20260816","name":"GLM 5.3","releaseLabel":"5.3","generationId":"5.3","generationLabel":"5.3","variantLabel":"Base","createdAt":"2026-08-18T20:57:35.000Z","kind":"language","contextLength":1048576,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":1.4,"completionPerMillion":4.4,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GLM-5.3 is a large-scale reasoning model from Z.ai, built for complex software engineering and long-horizon agent tasks. It supports text input and output with a 1M-token context window, and improves...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"z-ai/glm-5.3","url":"https://openrouter.ai/z-ai/glm-5.3","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":76.6,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"AA Intelligence Index","score":59.5,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/glm-5-3","measuredAt":"2026-08-27","scope":"Current public index; highest tested effort configuration"},{"benchmark":"AA Agentic Index","score":59.1,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/glm-5-3","measuredAt":"2026-08-27","scope":"Current public agentic composite"},{"benchmark":"SciCode","score":56.5,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/glm-5-3","measuredAt":"2026-08-27","scope":"Scientific coding evaluation"},{"benchmark":"AA Omniscience Index","score":14.3,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/glm-5-3","measuredAt":"2026-08-27","scope":"Current public knowledge and hallucination index"},{"benchmark":"MLCR overall","score":48.3,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/glm-5-3","measuredAt":"2026-08-27","scope":"Machine-learning code reasoning evaluation"},{"benchmark":"GDPval-AA normalized","score":63.1,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/glm-5-3","measuredAt":"2026-08-27","scope":"Normalized real-world knowledge-work evaluation"},{"benchmark":"τ²-bench Banking","score":50.3,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/glm-5-3","measuredAt":"2026-08-27","scope":"Tool-agent benchmark in the banking domain"},{"benchmark":"Terminal-Bench 2.1","score":83.9,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/glm-5-3","measuredAt":"2026-08-27","scope":"Terminal-agent task completion"},{"benchmark":"LCR","score":76.3,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/glm-5-3","measuredAt":"2026-08-27","scope":"Long-context reasoning evaluation"},{"benchmark":"Humanity's Last Exam","score":42.3,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/glm-5-3","measuredAt":"2026-08-27","scope":"Broad expert-level knowledge evaluation"},{"benchmark":"GPQA","score":91.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/glm-5-3","measuredAt":"2026-08-27","scope":"Graduate-level science questions"},{"benchmark":"CritPt","score":19.1,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/glm-5-3","measuredAt":"2026-08-27","scope":"Critical-point reasoning evaluation"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"a45ca9db-1953-484a-9426-cab823615321","title":"GLM-5.2","href":"/intel/a45ca9db-1953-484a-9426-cab823615321","hook":"Where open-weight models actually stand against the closed frontier on agentic coding, argued from GLM-5.2's benchmark results."}]},{"id":"z-ai/glm-5.3-flash-20260826","name":"GLM 5.3 Flash","releaseLabel":"5.3 Flash","generationId":"5.3","generationLabel":"5.3","variantLabel":"Flash","createdAt":"2026-08-26T13:59:01.000Z","kind":"multimodal","contextLength":1310720,"inputModalities":["text","image","video"],"outputModalities":["text"],"promptPerMillion":0.075,"completionPerMillion":0.25,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"GLM-5.3-Flash is a native multimodal model from Z.ai. It is suited for efficient coding and long-horizon agent tasks. Its hybrid sparse and linear attention architecture maintains accurate long-context behavior while...","huggingFaceId":"zai-org/GLM-5.3-Flash","routes":[{"provider":"OpenRouter","modelId":"z-ai/glm-5.3-flash","url":"https://openrouter.ai/z-ai/glm-5.3-flash","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":71.1,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"AA Intelligence Index","score":57.5,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/glm-5-3-flash","measuredAt":"2026-08-27","scope":"Current public index; highest tested effort configuration"},{"benchmark":"terminal-bench-2.1 · terminalbench 2 1","score":84.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3-Flash","measuredAt":"2026-08-26","scope":"Registered model evaluation · GLM-5.3-Flash model card"},{"benchmark":"deep-swe · deep swe","score":63.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3-Flash","measuredAt":"2026-08-26","scope":"Registered model evaluation · GLM-5.3-Flash model card"},{"benchmark":"hle · hle","score":55.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3-Flash","measuredAt":"2026-08-26","scope":"Registered model evaluation · GLM-5.3-Flash model card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"a45ca9db-1953-484a-9426-cab823615321","title":"GLM-5.2","href":"/intel/a45ca9db-1953-484a-9426-cab823615321","hook":"Where open-weight models actually stand against the closed frontier on agentic coding, argued from GLM-5.2's benchmark results."}]}]}]},{"id":"cohere","name":"Cohere","domain":"cohere.com","major":true,"modelCount":8,"series":[{"id":"command","label":"Command","models":[{"id":"cohere/command-r-08-2024","name":"Command R (08-2024)","releaseLabel":"R (08-2024)","generationId":"r","generationLabel":"R","variantLabel":"08 2024","createdAt":"2024-08-30T00:00:00.000Z","kind":"language","contextLength":128000,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.15,"completionPerMillion":0.6,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"command-r-08-2024 is an update of the [Command R](/models/cohere/command-r) with improved performance for multilingual retrieval-augmented generation (RAG) and tool use. More broadly, it is better at math, code and reasoning and...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"cohere/command-r-08-2024","url":"https://openrouter.ai/cohere/command-r-08-2024","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":32.6,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2025_04_25.csv","measuredAt":"2025-04-25","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1187,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 303 · 10,140 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"e9939137-ee00-4d13-94e3-314469e3c713","title":"Claude Code Best Practices Guide","href":"/intel/e9939137-ee00-4d13-94e3-314469e3c713","hook":"A dense, structured reference for the parts of Claude Code most people never touch — subagents, custom commands, skills, workflow composition. If your CLAUDE…"},{"id":"95dc1bae-18bf-410a-b8b6-accda66ab002","title":".claude Folder Configuration Guide","href":"/intel/95dc1bae-18bf-410a-b8b6-accda66ab002","hook":"The reference for everything that lives in .claude — CLAUDE.md, custom commands, permissions, project setup — in one place. The fastest way to turn a default…"},{"id":"cac69425-46f5-446b-9f5c-0d4716a483b5","title":"WTF Is a Loop? Part 2: The 15 Loops People Are Actually Running","href":"/intel/cac69425-46f5-446b-9f5c-0d4716a483b5","hook":"Matt Van Horn's breakdown of the 15 agent loops people actually run day to day, each with a copyable command. The practical sequel to his viral explainer — s…"}]},{"id":"cohere/command-r-plus-08-2024","name":"Command R+ (08-2024)","releaseLabel":"R+ (08-2024)","generationId":"r-plus","generationLabel":"R+","variantLabel":"08 2024","createdAt":"2024-08-30T00:00:00.000Z","kind":"language","contextLength":128000,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":2.5,"completionPerMillion":10,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"command-r-plus-08-2024 is an update of the [Command R+](/models/cohere/command-r-plus) with roughly 50% higher throughput and 25% lower latencies as compared to the previous Command R+ version, while keeping the hardware footprint...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"cohere/command-r-plus-08-2024","url":"https://openrouter.ai/cohere/command-r-plus-08-2024","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":32.6,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2025_04_25.csv","measuredAt":"2025-04-25","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1187,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 303 · 10,140 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"e9939137-ee00-4d13-94e3-314469e3c713","title":"Claude Code Best Practices Guide","href":"/intel/e9939137-ee00-4d13-94e3-314469e3c713","hook":"A dense, structured reference for the parts of Claude Code most people never touch — subagents, custom commands, skills, workflow composition. If your CLAUDE…"},{"id":"95dc1bae-18bf-410a-b8b6-accda66ab002","title":".claude Folder Configuration Guide","href":"/intel/95dc1bae-18bf-410a-b8b6-accda66ab002","hook":"The reference for everything that lives in .claude — CLAUDE.md, custom commands, permissions, project setup — in one place. The fastest way to turn a default…"},{"id":"cac69425-46f5-446b-9f5c-0d4716a483b5","title":"WTF Is a Loop? Part 2: The 15 Loops People Are Actually Running","href":"/intel/cac69425-46f5-446b-9f5c-0d4716a483b5","hook":"Matt Van Horn's breakdown of the 15 agent loops people actually run day to day, each with a copyable command. The practical sequel to his viral explainer — s…"}]},{"id":"cohere/command-r7b-12-2024","name":"Command R7B (12-2024)","releaseLabel":"R7B (12-2024)","generationId":"r7b","generationLabel":"R7b","variantLabel":"12 2024","createdAt":"2024-12-14T06:35:52.000Z","kind":"language","contextLength":128000,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.0375,"completionPerMillion":0.15,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Command R7B (12-2024) is a small, fast update of the Command R+ model, delivered in December 2024. It excels at RAG, tool use, agents, and similar tasks requiring complex reasoning...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"cohere/command-r7b-12-2024","url":"https://openrouter.ai/cohere/command-r7b-12-2024","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"e9939137-ee00-4d13-94e3-314469e3c713","title":"Claude Code Best Practices Guide","href":"/intel/e9939137-ee00-4d13-94e3-314469e3c713","hook":"A dense, structured reference for the parts of Claude Code most people never touch — subagents, custom commands, skills, workflow composition. If your CLAUDE…"},{"id":"95dc1bae-18bf-410a-b8b6-accda66ab002","title":".claude Folder Configuration Guide","href":"/intel/95dc1bae-18bf-410a-b8b6-accda66ab002","hook":"The reference for everything that lives in .claude — CLAUDE.md, custom commands, permissions, project setup — in one place. The fastest way to turn a default…"},{"id":"cac69425-46f5-446b-9f5c-0d4716a483b5","title":"WTF Is a Loop? Part 2: The 15 Loops People Are Actually Running","href":"/intel/cac69425-46f5-446b-9f5c-0d4716a483b5","hook":"Matt Van Horn's breakdown of the 15 agent loops people actually run day to day, each with a copyable command. The practical sequel to his viral explainer — s…"}]},{"id":"cohere/command-a-03-2025","name":"Command A","releaseLabel":"A","generationId":"a","generationLabel":"A","variantLabel":"Nd A","createdAt":"2025-03-13T19:32:22.000Z","kind":"language","contextLength":256000,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":2.5,"completionPerMillion":10,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Command A is an open-weights 111B parameter model with a 256k context window focused on delivering great performance across agentic, multilingual, and coding use cases. Compared to other leading proprietary...","huggingFaceId":"CohereForAI/c4ai-command-a-03-2025","routes":[{"provider":"OpenRouter","modelId":"cohere/command-a","url":"https://openrouter.ai/cohere/command-a","free":false,"batch":false}],"benchmarks":[{"benchmark":"AA Intelligence Index","score":22.8,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/command-a-plus","measuredAt":"2026-08-27","scope":"Current public index; highest tested effort configuration"},{"benchmark":"AA Agentic Index","score":9.2,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/command-a-plus","measuredAt":"2026-08-27","scope":"Current public agentic composite"},{"benchmark":"SciCode","score":37.8,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/command-a-plus","measuredAt":"2026-08-27","scope":"Scientific coding evaluation"},{"benchmark":"AA Omniscience Index","score":-4,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/command-a-plus","measuredAt":"2026-08-27","scope":"Current public knowledge and hallucination index"},{"benchmark":"MLCR overall","score":0.6,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/command-a-plus","measuredAt":"2026-08-27","scope":"Machine-learning code reasoning evaluation"},{"benchmark":"GDPval-AA normalized","score":10.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/command-a-plus","measuredAt":"2026-08-27","scope":"Normalized real-world knowledge-work evaluation"},{"benchmark":"τ²-bench","score":80.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/command-a-plus","measuredAt":"2026-08-27","scope":"Tool-agent benchmark across supported domains"},{"benchmark":"τ²-bench Banking","score":6,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/command-a-plus","measuredAt":"2026-08-27","scope":"Tool-agent benchmark in the banking domain"},{"benchmark":"Terminal-Bench Hard","score":25,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/command-a-plus","measuredAt":"2026-08-27","scope":"Hard terminal-agent tasks"},{"benchmark":"Terminal-Bench 2.1","score":22.8,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/command-a-plus","measuredAt":"2026-08-27","scope":"Terminal-agent task completion"},{"benchmark":"LCR","score":48.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/command-a-plus","measuredAt":"2026-08-27","scope":"Long-context reasoning evaluation"},{"benchmark":"IFBench","score":73.9,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/command-a-plus","measuredAt":"2026-08-27","scope":"Instruction-following evaluation"},{"benchmark":"Humanity's Last Exam","score":12,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/command-a-plus","measuredAt":"2026-08-27","scope":"Broad expert-level knowledge evaluation"},{"benchmark":"GPQA","score":76.1,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/command-a-plus","measuredAt":"2026-08-27","scope":"Graduate-level science questions"},{"benchmark":"CritPt","score":0.3,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/command-a-plus","measuredAt":"2026-08-27","scope":"Critical-point reasoning evaluation"},{"benchmark":"MMMU Pro","score":63.2,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/command-a-plus","measuredAt":"2026-08-27","scope":"Multimodal expert reasoning evaluation"},{"benchmark":"gpqa · diamond","score":50.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/b2/77/b27740dc-a017-4738-8dd0-c18973b61d9e.json","measuredAt":"2026-04-16","scope":"Registered model evaluation · EvalEval"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"e9939137-ee00-4d13-94e3-314469e3c713","title":"Claude Code Best Practices Guide","href":"/intel/e9939137-ee00-4d13-94e3-314469e3c713","hook":"A dense, structured reference for the parts of Claude Code most people never touch — subagents, custom commands, skills, workflow composition. If your CLAUDE…"},{"id":"95dc1bae-18bf-410a-b8b6-accda66ab002","title":".claude Folder Configuration Guide","href":"/intel/95dc1bae-18bf-410a-b8b6-accda66ab002","hook":"The reference for everything that lives in .claude — CLAUDE.md, custom commands, permissions, project setup — in one place. The fastest way to turn a default…"},{"id":"cac69425-46f5-446b-9f5c-0d4716a483b5","title":"WTF Is a Loop? Part 2: The 15 Loops People Are Actually Running","href":"/intel/cac69425-46f5-446b-9f5c-0d4716a483b5","hook":"Matt Van Horn's breakdown of the 15 agent loops people actually run day to day, each with a copyable command. The practical sequel to his viral explainer — s…"}]}]},{"id":"rerank","label":"Rerank","models":[{"id":"cohere/rerank-v3.5","name":"Rerank v3.5","releaseLabel":"v3.5","generationId":"v3.5","generationLabel":"V3.5","variantLabel":"Base","createdAt":"2026-04-05T19:09:18.000Z","kind":"language","contextLength":4096,"inputModalities":["text"],"outputModalities":["rerank"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Rerank v3.5 is designed to reorder search results for improved relevance. It supports multi-aspect and semi-structured data reranking over 100+ languages. Ideal for refining results from semantic or keyword search...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"cohere/rerank-v3.5","url":"https://openrouter.ai/cohere/rerank-v3.5","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"bc3e57e7-be1a-4585-a43a-df62724ccd70","title":"Latest open artifacts (#22): Zyphra, Cohere, and Poolside are expanding the breadth of the ecosystem","href":"/intel/bc3e57e7-be1a-4585-a43a-df62724ccd70","hook":"Most open-model coverage fixates on the frontier race, but this tracks a quieter structural shift: the release landscape has broadened from a handful of most…"}]},{"id":"cohere/rerank-4-fast","name":"Rerank 4 Fast","releaseLabel":"4 Fast","generationId":"4","generationLabel":"4","variantLabel":"Fast","createdAt":"2026-04-06T02:24:29.000Z","kind":"language","contextLength":32768,"inputModalities":["text"],"outputModalities":["rerank"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Cohere's AI search foundation model for enhancing the relevance of information surfaced within search and RAG systems. Features a 32K context window, multilingual support across 100+ languages, no data pre-processing...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"cohere/rerank-4-fast","url":"https://openrouter.ai/cohere/rerank-4-fast","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"bc3e57e7-be1a-4585-a43a-df62724ccd70","title":"Latest open artifacts (#22): Zyphra, Cohere, and Poolside are expanding the breadth of the ecosystem","href":"/intel/bc3e57e7-be1a-4585-a43a-df62724ccd70","hook":"Most open-model coverage fixates on the frontier race, but this tracks a quieter structural shift: the release landscape has broadened from a handful of most…"}]},{"id":"cohere/rerank-4-pro","name":"Rerank 4 Pro","releaseLabel":"4 Pro","generationId":"4","generationLabel":"4","variantLabel":"Pro","createdAt":"2026-04-06T03:30:47.000Z","kind":"language","contextLength":32768,"inputModalities":["text"],"outputModalities":["rerank"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Cohere's AI search foundation model for enhancing the relevance of information surfaced within search and RAG systems. Features a 32K context window, multilingual support across 100+ languages, no data pre-processing...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"cohere/rerank-4-pro","url":"https://openrouter.ai/cohere/rerank-4-pro","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"bc3e57e7-be1a-4585-a43a-df62724ccd70","title":"Latest open artifacts (#22): Zyphra, Cohere, and Poolside are expanding the breadth of the ecosystem","href":"/intel/bc3e57e7-be1a-4585-a43a-df62724ccd70","hook":"Most open-model coverage fixates on the frontier race, but this tracks a quieter structural shift: the release landscape has broadened from a handful of most…"}]}]},{"id":"north","label":"North","models":[{"id":"cohere/north-mini-code-20260617","name":"North Mini Code","releaseLabel":"Mini Code","generationId":"releases","generationLabel":"Releases","variantLabel":"Mini Code","createdAt":"2026-06-17T19:15:48.000Z","kind":"language","contextLength":256000,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"North Mini Code is Cohere's first agentic coding model and the debut of its North family. A sparse mixture-of-experts model with 30B total parameters and 3B active, it is optimized...","huggingFaceId":"CohereLabs/North-Mini-Code-1.0","routes":[{"provider":"OpenRouter","modelId":"cohere/north-mini-code:free","url":"https://openrouter.ai/cohere/north-mini-code:free","free":true,"batch":false}],"benchmarks":[{"benchmark":"SWE-bench_Pro · SWE Bench Pro","score":40.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/CohereLabs/North-Mini-Code-1.0","measuredAt":"2026-06-09","scope":"Registered model evaluation · North Mini Code Model Page"},{"benchmark":"SWE-bench_Verified · swe bench % resolved","score":67.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/CohereLabs/North-Mini-Code-1.0","measuredAt":"2026-06-09","scope":"Registered model evaluation · North Mini Code Model Page"},{"benchmark":"terminal-bench-2.0 · terminalbench 2","score":36,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/CohereLabs/North-Mini-Code-1.0","measuredAt":"2026-06-09","scope":"Registered model evaluation · North Mini Code Model Page"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"bc3e57e7-be1a-4585-a43a-df62724ccd70","title":"Latest open artifacts (#22): Zyphra, Cohere, and Poolside are expanding the breadth of the ecosystem","href":"/intel/bc3e57e7-be1a-4585-a43a-df62724ccd70","hook":"Most open-model coverage fixates on the frontier race, but this tracks a quieter structural shift: the release landscape has broadened from a handful of most…"}]}]}]},{"id":"nvidia","name":"NVIDIA","domain":"nvidia.com","major":true,"modelCount":11,"series":[{"id":"nemotron","label":"Nemotron","models":[{"id":"nvidia/nemotron-3-nano-30b-a3b","name":"Nemotron 3 Nano 30B A3B","releaseLabel":"3 Nano 30B A3B","generationId":"3","generationLabel":"3","variantLabel":"Nano 30B A3B","createdAt":"2025-12-14T16:54:35.000Z","kind":"language","contextLength":262144,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.049999999999999996,"completionPerMillion":0.19999999999999998,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...","huggingFaceId":"nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16","routes":[{"provider":"OpenRouter","modelId":"nvidia/nemotron-3-nano-30b-a3b","url":"https://openrouter.ai/nvidia/nemotron-3-nano-30b-a3b","free":false,"batch":false}],"benchmarks":[{"benchmark":"Humanity's Last Exam","score":15.5,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"MMLU-Pro","score":78.3,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"ifstruct-v1.0 · ifstruct v1","score":86.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://www.liquid.ai/blog/ifstruct-v1.0","measuredAt":"2026-06-30","scope":"Registered model evaluation · Liquid AI — IFStruct v1.0 blog (Nemotron-3-Nano-30B-A3B)"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"},{"id":"63baf567-7f04-4e96-92d9-86c3a615e53a","title":"#494 – Jensen Huang: NVIDIA &#8211; The $4 Trillion Company &#038; the AI Revolution","href":"/intel/63baf567-7f04-4e96-92d9-86c3a615e53a","hook":"Jensen Huang on rack-scale engineering, scaling laws and the supply-chain constraints that set the price of compute."}]},{"id":"nvidia/nemotron-3-super-120b-a12b-20230311","name":"Nemotron 3 Super","releaseLabel":"3 Super","generationId":"3","generationLabel":"3","variantLabel":"Super 120B A12B","createdAt":"2026-03-11T16:07:19.000Z","kind":"language","contextLength":1000000,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.08499999999999999,"completionPerMillion":0.39999999999999997,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"NVIDIA Nemotron 3 Super is a 120B-parameter open hybrid MoE model, activating just 12B parameters for maximum compute efficiency and accuracy in complex multi-agent applications. Built on a hybrid Mamba-Transformer...","huggingFaceId":"nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8","routes":[{"provider":"OpenRouter","modelId":"nvidia/nemotron-3-super-120b-a12b","url":"https://openrouter.ai/nvidia/nemotron-3-super-120b-a12b","free":false,"batch":false},{"provider":"OpenRouter","modelId":"nvidia/nemotron-3-super-120b-a12b:free","url":"https://openrouter.ai/nvidia/nemotron-3-super-120b-a12b:free","free":true,"batch":false}],"benchmarks":[{"benchmark":"gpqa · diamond","score":77.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/bd/1f/bd1fe04f-7d58-4045-be60-7150500d537f.json","measuredAt":"2026-04-16","scope":"Registered model evaluation · EvalEval"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"},{"id":"63baf567-7f04-4e96-92d9-86c3a615e53a","title":"#494 – Jensen Huang: NVIDIA &#8211; The $4 Trillion Company &#038; the AI Revolution","href":"/intel/63baf567-7f04-4e96-92d9-86c3a615e53a","hook":"Jensen Huang on rack-scale engineering, scaling laws and the supply-chain constraints that set the price of compute."}]},{"id":"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning-20260428","name":"Nemotron 3 Nano Omni","releaseLabel":"3 Nano Omni","generationId":"3","generationLabel":"3","variantLabel":"Nano Omni 30B A3B Reasoning","createdAt":"2026-04-28T16:18:15.000Z","kind":"multimodal","contextLength":256000,"inputModalities":["text","audio","image","video"],"outputModalities":["text"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"NVIDIA Nemotron™ 3 Nano Omni is a 30B-A3B open multimodal model designed to function as a perception and context sub-agent in enterprise agent systems. It accepts text, image, video, and...","huggingFaceId":"nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning-BF16","routes":[{"provider":"OpenRouter","modelId":"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free","url":"https://openrouter.ai/nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free","free":true,"batch":false}],"benchmarks":[{"benchmark":"ParseBench · mean","score":48.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/llamaindex/ParseBench","measuredAt":"2026-07-27","scope":"Registered model evaluation · ParseBench"},{"benchmark":"ParseBench · text content","score":81,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/llamaindex/ParseBench","measuredAt":"2026-07-27","scope":"Registered model evaluation · ParseBench"},{"benchmark":"ParseBench · text formatting","score":59.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/llamaindex/ParseBench","measuredAt":"2026-07-27","scope":"Registered model evaluation · ParseBench"},{"benchmark":"ParseBench · layout","score":0,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/llamaindex/ParseBench","measuredAt":"2026-07-27","scope":"Registered model evaluation · ParseBench"},{"benchmark":"ParseBench · chart","score":31.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/llamaindex/ParseBench","measuredAt":"2026-07-27","scope":"Registered model evaluation · ParseBench"},{"benchmark":"ParseBench · table","score":70.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/llamaindex/ParseBench","measuredAt":"2026-07-27","scope":"Registered model evaluation · ParseBench"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"},{"id":"63baf567-7f04-4e96-92d9-86c3a615e53a","title":"#494 – Jensen Huang: NVIDIA &#8211; The $4 Trillion Company &#038; the AI Revolution","href":"/intel/63baf567-7f04-4e96-92d9-86c3a615e53a","hook":"Jensen Huang on rack-scale engineering, scaling laws and the supply-chain constraints that set the price of compute."}]},{"id":"nvidia/nemotron-3-ultra-550b-a55b-20260604","name":"Nemotron 3 Ultra","releaseLabel":"3 Ultra","generationId":"3","generationLabel":"3","variantLabel":"Ultra 550B A55B","createdAt":"2026-06-04T05:33:28.000Z","kind":"language","contextLength":512288,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.6,"completionPerMillion":3.5999999999999996,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","huggingFaceId":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","routes":[{"provider":"OpenRouter","modelId":"nvidia/nemotron-3-ultra-550b-a55b","url":"https://openrouter.ai/nvidia/nemotron-3-ultra-550b-a55b","free":false,"batch":false},{"provider":"OpenRouter","modelId":"nvidia/nemotron-3-ultra-550b-a55b:batch","url":"https://openrouter.ai/nvidia/nemotron-3-ultra-550b-a55b:batch","free":false,"batch":true},{"provider":"OpenRouter","modelId":"nvidia/nemotron-3-ultra-550b-a55b:free","url":"https://openrouter.ai/nvidia/nemotron-3-ultra-550b-a55b:free","free":true,"batch":false}],"benchmarks":[{"benchmark":"AA Intelligence Index","score":38.3,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/nvidia-nemotron-3-ultra-550b-a55b","measuredAt":"2026-08-27","scope":"Current public index; highest tested effort configuration"},{"benchmark":"terminal-bench-2.1 · terminalbench 2 1","score":56.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","measuredAt":"2026-06-10","scope":"Registered model evaluation · Model Card"},{"benchmark":"SWE-bench_Multilingual · swe bench multilingual % resolved","score":67.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","measuredAt":"2026-08-10","scope":"Registered model evaluation · Model Card"},{"benchmark":"gpqa · diamond","score":87,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","measuredAt":"2026-06-03","scope":"Registered model evaluation · Model Card"},{"benchmark":"hle · hle","score":26.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","measuredAt":"2026-06-03","scope":"Registered model evaluation · Model Card"},{"benchmark":"MMLU-Pro · mmlu pro","score":86.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","measuredAt":"2026-06-03","scope":"Registered model evaluation · Model Card"},{"benchmark":"SWE-bench_Verified · swe bench % resolved","score":71.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","measuredAt":"2026-06-03","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"},{"id":"63baf567-7f04-4e96-92d9-86c3a615e53a","title":"#494 – Jensen Huang: NVIDIA &#8211; The $4 Trillion Company &#038; the AI Revolution","href":"/intel/63baf567-7f04-4e96-92d9-86c3a615e53a","hook":"Jensen Huang on rack-scale engineering, scaling laws and the supply-chain constraints that set the price of compute."}]},{"id":"nvidia/nemotron-3.5-lightning-20260807","name":"Nemotron 3.5 Lightning","releaseLabel":"3.5 Lightning","generationId":"3.5","generationLabel":"3.5","variantLabel":"Lightning","createdAt":"2026-08-11T12:52:31.000Z","kind":"language","contextLength":262144,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.08,"completionPerMillion":0.19999999999999998,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"NVIDIA Nemotron 3.5 Lightning is an open mixture-of-experts model from NVIDIA, with 3B active parameters out of 30B total. It is suited for high-throughput agentic workloads and specialized tasks that...","huggingFaceId":"nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16","routes":[{"provider":"OpenRouter","modelId":"nvidia/nemotron-3.5-lightning","url":"https://openrouter.ai/nvidia/nemotron-3.5-lightning","free":false,"batch":false},{"provider":"OpenRouter","modelId":"nvidia/nemotron-3.5-lightning:free","url":"https://openrouter.ai/nvidia/nemotron-3.5-lightning:free","free":true,"batch":false}],"benchmarks":[{"benchmark":"AA Intelligence Index","score":23.6,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/nemotron-3-5-lightning","measuredAt":"2026-08-27","scope":"Current public index; highest tested effort configuration"},{"benchmark":"AA Agentic Index","score":13.8,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/nemotron-3-5-lightning","measuredAt":"2026-08-27","scope":"Current public agentic composite"},{"benchmark":"SciCode","score":31.6,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/nemotron-3-5-lightning","measuredAt":"2026-08-27","scope":"Scientific coding evaluation"},{"benchmark":"AA Omniscience Index","score":-17.7,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/nemotron-3-5-lightning","measuredAt":"2026-08-27","scope":"Current public knowledge and hallucination index"},{"benchmark":"MLCR overall","score":1.1,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/nemotron-3-5-lightning","measuredAt":"2026-08-27","scope":"Machine-learning code reasoning evaluation"},{"benchmark":"GDPval-AA normalized","score":16.2,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/nemotron-3-5-lightning","measuredAt":"2026-08-27","scope":"Normalized real-world knowledge-work evaluation"},{"benchmark":"τ²-bench Banking","score":8.9,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/nemotron-3-5-lightning","measuredAt":"2026-08-27","scope":"Tool-agent benchmark in the banking domain"},{"benchmark":"Terminal-Bench 2.1","score":24.3,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/nemotron-3-5-lightning","measuredAt":"2026-08-27","scope":"Terminal-agent task completion"},{"benchmark":"LCR","score":55.3,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/nemotron-3-5-lightning","measuredAt":"2026-08-27","scope":"Long-context reasoning evaluation"},{"benchmark":"Humanity's Last Exam","score":10.6,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/nemotron-3-5-lightning","measuredAt":"2026-08-27","scope":"Broad expert-level knowledge evaluation"},{"benchmark":"GPQA","score":74.3,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/nemotron-3-5-lightning","measuredAt":"2026-08-27","scope":"Graduate-level science questions"},{"benchmark":"CritPt","score":0,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/nemotron-3-5-lightning","measuredAt":"2026-08-27","scope":"Critical-point reasoning evaluation"},{"benchmark":"Enterprise Ops Gym","score":18.4,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/nemotron-3-5-lightning","measuredAt":"2026-08-27","scope":"Enterprise operations agent tasks"},{"benchmark":"terminal-bench-2.1 · terminalbench 2 1","score":24.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16","measuredAt":"2026-08-13","scope":"Registered model evaluation · Model Card"},{"benchmark":"MMLU-Pro · mmlu pro","score":81.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16","measuredAt":"2026-08-11","scope":"Registered model evaluation · NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16 model card"},{"benchmark":"gpqa · diamond","score":75.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16","measuredAt":"2026-08-11","scope":"Registered model evaluation · NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16 model card"},{"benchmark":"hle · hle","score":11.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16","measuredAt":"2026-08-11","scope":"Registered model evaluation · NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16 model card"},{"benchmark":"SWE-bench_Verified · swe bench % resolved","score":51.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16","measuredAt":"2026-08-11","scope":"Registered model evaluation · NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16 model card"},{"benchmark":"SWE-bench_Multilingual · swe bench multilingual % resolved","score":39.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16","measuredAt":"2026-08-11","scope":"Registered model evaluation · NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16 model card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"},{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"},{"id":"63baf567-7f04-4e96-92d9-86c3a615e53a","title":"#494 – Jensen Huang: NVIDIA &#8211; The $4 Trillion Company &#038; the AI Revolution","href":"/intel/63baf567-7f04-4e96-92d9-86c3a615e53a","hook":"Jensen Huang on rack-scale engineering, scaling laws and the supply-chain constraints that set the price of compute."}]}]},{"id":"llama-nemotron-embed","label":"Llama Nemotron Embed","models":[{"id":"nvidia/llama-nemotron-embed-vl-1b-v2-20260224","name":"Llama Nemotron Embed VL 1B V2","releaseLabel":"VL 1B V2","generationId":"1b","generationLabel":"1B","variantLabel":"V2","createdAt":"2026-02-25T18:43:37.000Z","kind":"embedding","contextLength":131072,"inputModalities":["text","image"],"outputModalities":["embeddings"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"The Llama Nemotron Embed VL 1B V2 embedding model is optimized for multimodal question-answering retrieval. The model can embed 'documents' in the form of image, text, or image and text...","huggingFaceId":"nvidia/llama-nemotron-embed-vl-1b-v2","routes":[{"provider":"OpenRouter","modelId":"nvidia/llama-nemotron-embed-vl-1b-v2:free","url":"https://openrouter.ai/nvidia/llama-nemotron-embed-vl-1b-v2:free","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"},{"id":"63baf567-7f04-4e96-92d9-86c3a615e53a","title":"#494 – Jensen Huang: NVIDIA &#8211; The $4 Trillion Company &#038; the AI Revolution","href":"/intel/63baf567-7f04-4e96-92d9-86c3a615e53a","hook":"Jensen Huang on rack-scale engineering, scaling laws and the supply-chain constraints that set the price of compute."}]}]},{"id":"llama-nemotron-rerank","label":"Llama Nemotron Rerank","models":[{"id":"nvidia/llama-nemotron-rerank-vl-1b-v2","name":"Llama Nemotron Rerank VL 1B V2","releaseLabel":"VL 1B V2","generationId":"1b","generationLabel":"1B","variantLabel":"V2","createdAt":"2026-06-09T20:14:14.000Z","kind":"multimodal","contextLength":10240,"inputModalities":["text","image"],"outputModalities":["rerank"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Llama Nemotron Rerank VL 1B V2 is a 1.7B multimodal reranking model from NVIDIA. It evaluates the relevance of document images and text against user queries, designed for vision RAG...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"nvidia/llama-nemotron-rerank-vl-1b-v2:free","url":"https://openrouter.ai/nvidia/llama-nemotron-rerank-vl-1b-v2:free","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"},{"id":"63baf567-7f04-4e96-92d9-86c3a615e53a","title":"#494 – Jensen Huang: NVIDIA &#8211; The $4 Trillion Company &#038; the AI Revolution","href":"/intel/63baf567-7f04-4e96-92d9-86c3a615e53a","hook":"Jensen Huang on rack-scale engineering, scaling laws and the supply-chain constraints that set the price of compute."}]}]},{"id":"nemotron-asr","label":"Nemotron ASR","models":[{"id":"nvidia/nemotron-3.5-asr-streaming-multilingual-0.6b-20260813","name":"Nemotron 3.5 ASR Streaming Multilingual 0.6B","releaseLabel":"Nemotron 3.5 ASR Streaming Multilingual 0.6B","generationId":"3.5","generationLabel":"3.5","variantLabel":"Streaming Multilingual 0.6B","createdAt":"2026-08-13T20:52:51.000Z","kind":"audio","contextLength":0,"inputModalities":["audio"],"outputModalities":["transcription"],"promptPerMillion":3.33,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Nemotron 3.5 ASR Streaming Multilingual 0.6B is a speech recognition model from NVIDIA. Its prompt-conditioned, cache-aware FastConformer-RNNT design targets low-latency transcription across more than 40 languages for real-time captioning, voice...","huggingFaceId":"nvidia/Nemotron-3.5-ASR-Streaming-Multilingual-0.6b","routes":[{"provider":"OpenRouter","modelId":"nvidia/nemotron-3.5-asr-streaming-multilingual-0.6b","url":"https://openrouter.ai/nvidia/nemotron-3.5-asr-streaming-multilingual-0.6b","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"},{"id":"63baf567-7f04-4e96-92d9-86c3a615e53a","title":"#494 – Jensen Huang: NVIDIA &#8211; The $4 Trillion Company &#038; the AI Revolution","href":"/intel/63baf567-7f04-4e96-92d9-86c3a615e53a","hook":"Jensen Huang on rack-scale engineering, scaling laws and the supply-chain constraints that set the price of compute."}]}]},{"id":"nemotron-embed","label":"Nemotron Embed","models":[{"id":"nvidia/nemotron-3-embed-1b-20260716","name":"Nemotron 3 Embed 1B","releaseLabel":"Nemotron 3 Embed 1B","generationId":"3","generationLabel":"3","variantLabel":"1B","createdAt":"2026-07-16T12:01:34.000Z","kind":"embedding","contextLength":32768,"inputModalities":["text"],"outputModalities":["embeddings"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"NVIDIA Nemotron 3 Embed 1B is an open text embedding model from NVIDIA, optimized for high-throughput, low-latency retrieval. It is suited for enterprise search, RAG, code retrieval, and agentic retrieval...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"nvidia/nemotron-3-embed-1b:free","url":"https://openrouter.ai/nvidia/nemotron-3-embed-1b:free","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"},{"id":"63baf567-7f04-4e96-92d9-86c3a615e53a","title":"#494 – Jensen Huang: NVIDIA &#8211; The $4 Trillion Company &#038; the AI Revolution","href":"/intel/63baf567-7f04-4e96-92d9-86c3a615e53a","hook":"Jensen Huang on rack-scale engineering, scaling laws and the supply-chain constraints that set the price of compute."}]}]},{"id":"nemotron-safety","label":"Nemotron Safety","models":[{"id":"nvidia/nemotron-3.5-content-safety-20260604","name":"Nemotron 3.5 Content Safety","releaseLabel":"Nemotron 3.5 Content Safety","generationId":"3.5","generationLabel":"3.5","variantLabel":"Base","createdAt":"2026-06-04T14:04:24.000Z","kind":"multimodal","contextLength":128000,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":true,"description":"NVIDIA Nemotron 3.5 Content Safety is a compact 4B-parameter multimodal guardrail model from NVIDIA, fine-tuned from Google Gemma-3-4B. It moderates both inputs to and responses from LLMs and VLMs, accepting...","huggingFaceId":"nvidia/Nemotron-3.5-Content-Safety","routes":[{"provider":"OpenRouter","modelId":"nvidia/nemotron-3.5-content-safety:free","url":"https://openrouter.ai/nvidia/nemotron-3.5-content-safety:free","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"},{"id":"63baf567-7f04-4e96-92d9-86c3a615e53a","title":"#494 – Jensen Huang: NVIDIA &#8211; The $4 Trillion Company &#038; the AI Revolution","href":"/intel/63baf567-7f04-4e96-92d9-86c3a615e53a","hook":"Jensen Huang on rack-scale engineering, scaling laws and the supply-chain constraints that set the price of compute."}]}]},{"id":"parakeet","label":"Parakeet","models":[{"id":"nvidia/parakeet-tdt-0.6b-v3","name":"Parakeet TDT 0.6B v3","releaseLabel":"TDT 0.6B v3","generationId":"0.6b","generationLabel":"0.6B","variantLabel":"V3","createdAt":"2026-05-27T02:18:55.000Z","kind":"audio","contextLength":0,"inputModalities":["audio"],"outputModalities":["transcription"],"promptPerMillion":1500,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Parakeet TDT 0.6B v3 is NVIDIA's 600M-parameter multilingual speech-to-text model built on the FastConformer-TDT architecture. Trained on the Granary dataset (670,000+ hours of audio), it supports automatic language detection across...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"nvidia/parakeet-tdt-0.6b-v3","url":"https://openrouter.ai/nvidia/parakeet-tdt-0.6b-v3","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[{"id":"2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","title":"#459 – DeepSeek, China, OpenAI, NVIDIA, xAI, TSMC, Stargate, and AI Megaclusters","href":"/intel/2f54c26f-6d0f-4e1c-ada0-79ec76dbd5c6","hook":"A deep technical breakdown of how DeepSeek achieved frontier-level models at dramatically lower training cost, plus the GPU supply/export-control realities s…"},{"id":"63baf567-7f04-4e96-92d9-86c3a615e53a","title":"#494 – Jensen Huang: NVIDIA &#8211; The $4 Trillion Company &#038; the AI Revolution","href":"/intel/63baf567-7f04-4e96-92d9-86c3a615e53a","hook":"Jensen Huang on rack-scale engineering, scaling laws and the supply-chain constraints that set the price of compute."}]}]}]},{"id":"recraft","name":"Recraft","domain":"recraft.ai","major":false,"modelCount":15,"series":[{"id":"recraft","label":"Recraft","models":[{"id":"recraft/recraft-v3-20260413","name":"Recraft V3","releaseLabel":"V3","generationId":"v3","generationLabel":"V3","variantLabel":"Base","createdAt":"2026-05-07T20:23:53.000Z","kind":"image","contextLength":65536,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":9.580838323353289,"supportsTools":false,"supportsReasoning":false,"description":"Recraft V3 is an image generation model from Recraft. It supports text and image inputs with image output at ~1K resolution across multiple aspect ratios. Supports the following `image_config` parameters:...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"recraft/recraft-v3","url":"https://openrouter.ai/recraft/recraft-v3","free":true,"batch":false}],"benchmarks":[{"benchmark":"AA Image Arena Elo","score":1078,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/image/leaderboard/text-to-image","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[]},{"id":"recraft/recraft-v4-20260413","name":"Recraft V4","releaseLabel":"V4","generationId":"v4","generationLabel":"V4","variantLabel":"Base","createdAt":"2026-05-07T20:23:57.000Z","kind":"image","contextLength":65536,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":9.580838323353289,"supportsTools":false,"supportsReasoning":false,"description":"Recraft V4 is an image generation model from Recraft. It supports text and image inputs with image output at ~1K resolution across multiple aspect ratios. It delivers stronger compositional judgment,...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"recraft/recraft-v4","url":"https://openrouter.ai/recraft/recraft-v4","free":true,"batch":false}],"benchmarks":[{"benchmark":"AA Image Arena Elo","score":1180,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/image/leaderboard/text-to-image","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[]},{"id":"recraft/recraft-v4-pro-20260413","name":"Recraft V4 Pro","releaseLabel":"V4 Pro","generationId":"v4","generationLabel":"V4","variantLabel":"Pro","createdAt":"2026-05-07T20:24:01.000Z","kind":"image","contextLength":65536,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":59.8802395209581,"supportsTools":false,"supportsReasoning":false,"description":"Recraft V4 Pro is an image generation model from Recraft. It supports text and image inputs with image output at ~2K resolution across multiple aspect ratios, double the resolution of...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"recraft/recraft-v4-pro","url":"https://openrouter.ai/recraft/recraft-v4-pro","free":true,"batch":false}],"benchmarks":[{"benchmark":"AA Image Arena Elo","score":1194,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/image/leaderboard/text-to-image","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[]},{"id":"recraft/recraft-v4-vector-20260514","name":"Recraft V4 Vector","releaseLabel":"V4 Vector","generationId":"v4","generationLabel":"V4","variantLabel":"Vector","createdAt":"2026-05-13T21:22:13.000Z","kind":"image","contextLength":65536,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":19.1616766467066,"supportsTools":false,"supportsReasoning":false,"description":"Recraft V4 Vector is the vector (SVG) variant of Recraft V4. It supports text and image inputs and produces vector image output across multiple aspect ratios. Compared to the raster...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"recraft/recraft-v4-vector","url":"https://openrouter.ai/recraft/recraft-v4-vector","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[]},{"id":"recraft/recraft-v4-pro-vector-20260514","name":"Recraft V4 Pro Vector","releaseLabel":"V4 Pro Vector","generationId":"v4","generationLabel":"V4","variantLabel":"Pro Vector","createdAt":"2026-05-13T21:22:14.000Z","kind":"image","contextLength":65536,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":71.8562874251497,"supportsTools":false,"supportsReasoning":false,"description":"Recraft V4 Pro Vector is the vector (SVG) variant of Recraft V4 Pro. It supports text and image inputs and produces vector image output across multiple aspect ratios at the...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"recraft/recraft-v4-pro-vector","url":"https://openrouter.ai/recraft/recraft-v4-pro-vector","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[]},{"id":"recraft/recraft-v4.1-20260514","name":"Recraft V4.1","releaseLabel":"V4.1","generationId":"v4.1","generationLabel":"V4.1","variantLabel":"Base","createdAt":"2026-05-13T21:23:01.000Z","kind":"image","contextLength":65536,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":8.38323353293413,"supportsTools":false,"supportsReasoning":false,"description":"Recraft V4.1 is an image generation model from Recraft tuned for high aesthetics. It supports text and image inputs with image output at ~1K resolution across multiple aspect ratios, with...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"recraft/recraft-v4.1","url":"https://openrouter.ai/recraft/recraft-v4.1","free":true,"batch":false}],"benchmarks":[{"benchmark":"AA Image Arena Elo","score":1199,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/image/leaderboard/text-to-image","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[]},{"id":"recraft/recraft-v4.1-pro-20260514","name":"Recraft V4.1 Pro","releaseLabel":"V4.1 Pro","generationId":"v4.1","generationLabel":"V4.1","variantLabel":"Pro","createdAt":"2026-05-13T21:23:04.000Z","kind":"image","contextLength":65536,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":50.2994011976048,"supportsTools":false,"supportsReasoning":false,"description":"Recraft V4.1 Pro is an image generation model from Recraft tuned for high aesthetics. It supports text and image inputs with image output at ~2K resolution across multiple aspect ratios...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"recraft/recraft-v4.1-pro","url":"https://openrouter.ai/recraft/recraft-v4.1-pro","free":true,"batch":false}],"benchmarks":[{"benchmark":"AA Image Arena Elo","score":1189,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/image/leaderboard/text-to-image","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[]},{"id":"recraft/recraft-v4.1-utility-20260514","name":"Recraft V4.1 Utility","releaseLabel":"V4.1 Utility","generationId":"v4.1","generationLabel":"V4.1","variantLabel":"Utility","createdAt":"2026-05-13T21:23:07.000Z","kind":"image","contextLength":65536,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":8.38323353293413,"supportsTools":false,"supportsReasoning":false,"description":"Recraft V4.1 Utility is a general-purpose image generation model from Recraft. It supports text and image inputs with image output at ~1K resolution across multiple aspect ratios, with typical generation...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"recraft/recraft-v4.1-utility","url":"https://openrouter.ai/recraft/recraft-v4.1-utility","free":true,"batch":false}],"benchmarks":[{"benchmark":"AA Image Arena Elo","score":1217,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/image/leaderboard/text-to-image","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[]},{"id":"recraft/recraft-v4.1-utility-pro-20260514","name":"Recraft V4.1 Utility Pro","releaseLabel":"V4.1 Utility Pro","generationId":"v4.1","generationLabel":"V4.1","variantLabel":"Utility Pro","createdAt":"2026-05-13T21:23:09.000Z","kind":"image","contextLength":65536,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":50.2994011976048,"supportsTools":false,"supportsReasoning":false,"description":"Recraft V4.1 Utility Pro is a general-purpose image generation model from Recraft. It supports text and image inputs with image output at ~2K resolution across multiple aspect ratios — double...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"recraft/recraft-v4.1-utility-pro","url":"https://openrouter.ai/recraft/recraft-v4.1-utility-pro","free":true,"batch":false}],"benchmarks":[{"benchmark":"AA Image Arena Elo","score":1220,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/image/leaderboard/text-to-image","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[]},{"id":"recraft/recraft-v4.1-vector-20260514","name":"Recraft V4.1 Vector","releaseLabel":"V4.1 Vector","generationId":"v4.1","generationLabel":"V4.1","variantLabel":"Vector","createdAt":"2026-05-13T21:23:12.000Z","kind":"image","contextLength":65536,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":19.1616766467066,"supportsTools":false,"supportsReasoning":false,"description":"Recraft V4.1 Vector is the vector (SVG) variant of Recraft V4.1, tuned for high aesthetics. It supports text and image inputs and produces SVG image output across multiple aspect ratios,...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"recraft/recraft-v4.1-vector","url":"https://openrouter.ai/recraft/recraft-v4.1-vector","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[]},{"id":"recraft/recraft-v4.1-pro-vector-20260514","name":"Recraft V4.1 Pro Vector","releaseLabel":"V4.1 Pro Vector","generationId":"v4.1","generationLabel":"V4.1","variantLabel":"Pro Vector","createdAt":"2026-05-13T21:23:15.000Z","kind":"image","contextLength":65536,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":71.8562874251497,"supportsTools":false,"supportsReasoning":false,"description":"Recraft V4.1 Pro Vector is the vector (SVG) variant of Recraft V4.1 Pro, tuned for high aesthetics. It supports text and image inputs and produces higher-resolution SVG image output across...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"recraft/recraft-v4.1-pro-vector","url":"https://openrouter.ai/recraft/recraft-v4.1-pro-vector","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[]},{"id":"recraft/recraft-v4-styles-20260826","name":"Recraft V4 Styles","releaseLabel":"V4 Styles","generationId":"v4","generationLabel":"V4","variantLabel":"Styles","createdAt":"2026-08-26T11:07:38.000Z","kind":"image","contextLength":65536,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":8.38323353293413,"supportsTools":false,"supportsReasoning":false,"description":"Recraft V4 Styles is a style-consistent image generation model from Recraft. Every request requires at least one style reference image and generates a new image that reproduces the reference's rendering...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"recraft/recraft-v4-styles","url":"https://openrouter.ai/recraft/recraft-v4-styles","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[]},{"id":"recraft/recraft-v4-styles-pro-vector-20260826","name":"Recraft V4 Styles Pro Vector","releaseLabel":"V4 Styles Pro Vector","generationId":"v4","generationLabel":"V4","variantLabel":"Styles Pro Vector","createdAt":"2026-08-26T11:10:28.000Z","kind":"image","contextLength":65536,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":28.742514970059897,"supportsTools":false,"supportsReasoning":false,"description":"Recraft V4 Styles Pro Vector is a style-consistent image generation model from Recraft. Every request requires at least one style reference image and generates a new image that reproduces the...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"recraft/recraft-v4-styles-pro-vector","url":"https://openrouter.ai/recraft/recraft-v4-styles-pro-vector","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[]},{"id":"recraft/recraft-v4-styles-vector-20260826","name":"Recraft V4 Styles Vector","releaseLabel":"V4 Styles Vector","generationId":"v4","generationLabel":"V4","variantLabel":"Styles Vector","createdAt":"2026-08-26T11:10:35.000Z","kind":"image","contextLength":65536,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":11.9760479041916,"supportsTools":false,"supportsReasoning":false,"description":"Recraft V4 Styles Vector is a style-consistent image generation model from Recraft. Every request requires at least one style reference image and generates a new image that reproduces the reference's...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"recraft/recraft-v4-styles-vector","url":"https://openrouter.ai/recraft/recraft-v4-styles-vector","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[]},{"id":"recraft/recraft-v4-styles-pro-20260826","name":"Recraft V4 Styles Pro","releaseLabel":"V4 Styles Pro","generationId":"v4","generationLabel":"V4","variantLabel":"Styles Pro","createdAt":"2026-08-26T11:10:40.000Z","kind":"image","contextLength":65536,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":23.9520958083832,"supportsTools":false,"supportsReasoning":false,"description":"Recraft V4 Styles Pro is a style-consistent image generation model from Recraft. Every request requires at least one style reference image and generates a new image that reproduces the reference's...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"recraft/recraft-v4-styles-pro","url":"https://openrouter.ai/recraft/recraft-v4-styles-pro","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[]}]}]},{"id":"minimax","name":"MiniMax","domain":"minimax.io","major":false,"modelCount":12,"series":[{"id":"minimax","label":"Minimax","models":[{"id":"minimax/minimax-01","name":"MiniMax-01","releaseLabel":"01","generationId":"01","generationLabel":"01","variantLabel":"Base","createdAt":"2025-01-15T04:31:02.000Z","kind":"multimodal","contextLength":1000192,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0.19999999999999998,"completionPerMillion":1.1,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"MiniMax-01 is a combines MiniMax-Text-01 for text generation and MiniMax-VL-01 for image understanding. It has 456 billion parameters, with 45.9 billion parameters activated per inference, and can handle a context...","huggingFaceId":"MiniMaxAI/MiniMax-Text-01","routes":[{"provider":"OpenRouter","modelId":"minimax/minimax-01","url":"https://openrouter.ai/minimax/minimax-01","free":false,"batch":false}],"benchmarks":[{"benchmark":"MMLU-Pro · mmlu pro","score":75.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/2a/73/2a734d0f-7fcf-48d0-bea6-ad17cd749ee4.json","measuredAt":"2026-06-30","scope":"Registered model evaluation · EvalEval"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"minimax/minimax-m1","name":"MiniMax M1","releaseLabel":"M1","generationId":"releases","generationLabel":"Releases","variantLabel":"M1","createdAt":"2025-06-17T22:46:54.000Z","kind":"language","contextLength":1000000,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.55,"completionPerMillion":2.2,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"MiniMax-M1 is a large-scale, open-weight reasoning model designed for extended context and high-efficiency inference. It leverages a hybrid Mixture-of-Experts (MoE) architecture paired with a custom \"lightning attention\" mechanism, allowing it...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"minimax/minimax-m1","url":"https://openrouter.ai/minimax/minimax-m1","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1342,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 191 · 35,247 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"minimax/minimax-m2","name":"MiniMax M2","releaseLabel":"M2","generationId":"releases","generationLabel":"Releases","variantLabel":"M2","createdAt":"2025-10-23T20:41:33.000Z","kind":"language","contextLength":204800,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.255,"completionPerMillion":1.02,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"MiniMax-M2 is a compact, high-efficiency large language model optimized for end-to-end coding and agentic workflows. With 10 billion activated parameters (230 billion total), it delivers near-frontier intelligence across general reasoning,...","huggingFaceId":"MiniMaxAI/MiniMax-M2","routes":[{"provider":"OpenRouter","modelId":"minimax/minimax-m2","url":"https://openrouter.ai/minimax/minimax-m2","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":65.3,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2025_05_30.csv","measuredAt":"2025-05-30","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1342,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 190 · 6,903 votes"},{"benchmark":"Humanity's Last Exam","score":12.5,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"MMLU-Pro","score":82,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"SWE-bench Verified","score":69.4,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"Terminal-Bench","score":30,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"SWE-bench_Verified · swe bench % resolved","score":69.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M2","measuredAt":"2026-03-17","scope":"Registered model evaluation · Model Card"},{"benchmark":"terminal-bench-2.0 · terminalbench 2","score":30,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://www.tbench.ai/leaderboard/terminal-bench/2.0","measuredAt":"2025-11-01","scope":"Registered model evaluation · Terminal-Bench Leaderboard"},{"benchmark":"hle · hle","score":12.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M2","measuredAt":"2026-01-28","scope":"Registered model evaluation · Model Card"},{"benchmark":"MMLU-Pro · mmlu pro","score":82,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M2","measuredAt":"2026-01-28","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"minimax/minimax-m2.1","name":"MiniMax M2.1","releaseLabel":"M2.1","generationId":"releases","generationLabel":"Releases","variantLabel":"M2.1","createdAt":"2025-12-23T01:56:37.000Z","kind":"language","contextLength":204800,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.3,"completionPerMillion":1.2,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"MiniMax-M2.1 is a lightweight, state-of-the-art large language model optimized for coding, agentic workflows, and modern application development. With only 10 billion activated parameters, it delivers a major jump in real-world...","huggingFaceId":"MiniMaxAI/MiniMax-M2.1","routes":[{"provider":"OpenRouter","modelId":"minimax/minimax-m2.1","url":"https://openrouter.ai/minimax/minimax-m2.1","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1391,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 148 · 17,151 votes"},{"benchmark":"EvasionBench","score":71.3,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"Humanity's Last Exam","score":22.2,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"MMLU-Pro","score":88,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"SWE-bench Pro","score":36.8,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"SWE-bench Verified","score":74,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"Terminal-Bench","score":29.2,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"gpqa · diamond","score":80.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/08/7b/087b0dc6-3c87-4f40-8458-16f19de92200.json","measuredAt":"2026-04-17","scope":"Registered model evaluation · EvalEval"},{"benchmark":"MMLU-Pro · mmlu pro","score":88,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/dd/c9/ddc99cec-f5d3-4cc4-9cc5-bb534377b5f6.json","measuredAt":"2026-06-30","scope":"Registered model evaluation · EvalEval"},{"benchmark":"SWE-bench_Verified · swe bench % resolved","score":74,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M2.1","measuredAt":"2026-03-17","scope":"Registered model evaluation · Model Card"},{"benchmark":"SWE-bench_Pro · SWE Bench Pro","score":36.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://scale.com/leaderboard/swe_bench_pro_public","measuredAt":"2026-02-28","scope":"Registered model evaluation · SWE-Bench Pro official evaluation results"},{"benchmark":"terminal-bench-2.0 · terminalbench 2","score":29.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://www.tbench.ai/leaderboard/terminal-bench/2.0","measuredAt":"2025-12-23","scope":"Registered model evaluation · Terminal-Bench Leaderboard"},{"benchmark":"EvasionBench · evasion bench","score":71.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://arxiv.org/abs/2601.09142","measuredAt":"2026-02-10","scope":"Registered model evaluation · EvasionBench Paper"},{"benchmark":"hle · hle","score":22.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M2.1","measuredAt":"2026-01-14","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"minimax/minimax-m2-her-20260123","name":"MiniMax M2-her","releaseLabel":"M2-her","generationId":"releases","generationLabel":"Releases","variantLabel":"M2 Her","createdAt":"2026-01-23T14:07:19.000Z","kind":"language","contextLength":65536,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.3,"completionPerMillion":1.2,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"MiniMax M2-her is a dialogue-first large language model built for immersive roleplay, character-driven chat, and expressive multi-turn conversations. Designed to stay consistent in tone and personality, it supports rich message...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"minimax/minimax-m2-her","url":"https://openrouter.ai/minimax/minimax-m2-her","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"minimax/minimax-m2.5-20260211","name":"MiniMax M2.5","releaseLabel":"M2.5","generationId":"releases","generationLabel":"Releases","variantLabel":"M2.5","createdAt":"2026-02-12T15:01:42.000Z","kind":"language","contextLength":204800,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.27,"completionPerMillion":1.08,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...","huggingFaceId":"MiniMaxAI/MiniMax-M2.5","routes":[{"provider":"OpenRouter","modelId":"minimax/minimax-m2.5","url":"https://openrouter.ai/minimax/minimax-m2.5","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":60.3,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1359,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 178 · 41,032 votes"},{"benchmark":"GPQA","score":85.2,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"Humanity's Last Exam","score":19.4,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"SWE-bench Pro","score":55.4,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"SWE-bench Verified","score":75.8,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"gpqa · diamond","score":85.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M2.5","measuredAt":"2026-02-13","scope":"Registered model evaluation · Model Card"},{"benchmark":"hle · hle","score":19.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M2.5","measuredAt":"2026-02-13","scope":"Registered model evaluation · Model Card"},{"benchmark":"SWE-bench_Verified · swe bench % resolved","score":75.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://www.swebench.com/","measuredAt":"2026-03-10","scope":"Registered model evaluation · SWE-Bench official evaluation"},{"benchmark":"MMLU-Pro · mmlu pro","score":80.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/c5/4c/c54c4ee8-ff99-4cda-a81f-a2e3a4347fb8.json","measuredAt":"2026-06-30","scope":"Registered model evaluation · EvalEval"},{"benchmark":"WildClawBench · overall","score":27.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://internlm.github.io/WildClawBench","measuredAt":"2026-05-22","scope":"Registered model evaluation · WildClawBench"},{"benchmark":"apex-agents · apex-agents","score":6.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://www.mercor.com/apex/apex-agents-leaderboard/","measuredAt":"2026-04-22","scope":"Registered model evaluation · APEX-Agents Leaderboard"},{"benchmark":"SWE-bench_Pro · SWE Bench Pro","score":55.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M2.5","measuredAt":"2026-03-18","scope":"Registered model evaluation · Model card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"minimax/minimax-m2.7-20260318","name":"MiniMax M2.7","releaseLabel":"M2.7","generationId":"releases","generationLabel":"Releases","variantLabel":"M2.7","createdAt":"2026-03-18T12:24:57.000Z","kind":"language","contextLength":204800,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.3,"completionPerMillion":1.2,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...","huggingFaceId":"MiniMaxAI/MiniMax-M2.7","routes":[{"provider":"OpenRouter","modelId":"minimax/minimax-m2.7","url":"https://openrouter.ai/minimax/minimax-m2.7","free":false,"batch":false},{"provider":"OpenRouter","modelId":"minimax/minimax-m2.7:free","url":"https://openrouter.ai/minimax/minimax-m2.7:free","free":true,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":65,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_01_08.csv","measuredAt":"2026-01-08","scope":"Mean across that public release's task columns"},{"benchmark":"LMArena overall","score":1405,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 132 · 62,076 votes"},{"benchmark":"SWE-bench_Multilingual · swe bench multilingual % resolved","score":76.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M2.7","measuredAt":"2026-08-10","scope":"Registered model evaluation · Model Card"},{"benchmark":"skillsbench · skillsbench v1 1","score":34.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/benchflow/skillsbench-leaderboard/raw/main/leaderboard/skillsbench/v1.1/official.json","measuredAt":"2026-06-11","scope":"Registered model evaluation · SkillsBench v1.1 official leaderboard"},{"benchmark":"WildClawBench · overall","score":33.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://internlm.github.io/WildClawBench","measuredAt":"2026-05-22","scope":"Registered model evaluation · WildClawBench"},{"benchmark":"Claw-Eval · general","score":49.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://claw-eval.github.io","measuredAt":"2026-04-23","scope":"Registered model evaluation · Claw-Eval Leaderboard"},{"benchmark":"Claw-Eval · multi turn","score":44.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://claw-eval.github.io","measuredAt":"2026-04-23","scope":"Registered model evaluation · Claw-Eval Leaderboard"},{"benchmark":"SWE-bench_Pro · SWE Bench Pro","score":56.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M2.7","measuredAt":"2026-04-12","scope":"Registered model evaluation · Model Card"},{"benchmark":"terminal-bench-2.0 · terminalbench 2","score":57,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M2.7","measuredAt":"2026-04-12","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"minimax/minimax-m3-20260531","name":"MiniMax M3","releaseLabel":"M3","generationId":"releases","generationLabel":"Releases","variantLabel":"M3","createdAt":"2026-05-31T16:36:14.000Z","kind":"multimodal","contextLength":1048576,"inputModalities":["text","image","video"],"outputModalities":["text"],"promptPerMillion":0.3,"completionPerMillion":1.2,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...","huggingFaceId":"MiniMaxAI/Minimax-M3","routes":[{"provider":"OpenRouter","modelId":"minimax/minimax-m3","url":"https://openrouter.ai/minimax/minimax-m3","free":false,"batch":false},{"provider":"OpenRouter","modelId":"minimax/minimax-m3:batch","url":"https://openrouter.ai/minimax/minimax-m3:batch","free":false,"batch":true},{"provider":"OpenRouter","modelId":"minimax/minimax-m3:free","url":"https://openrouter.ai/minimax/minimax-m3:free","free":true,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":67.5,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"AA Intelligence Index","score":45.4,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/minimax-m3","measuredAt":"2026-08-27","scope":"Current public index; highest tested effort configuration"},{"benchmark":"LMArena overall","score":1435,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 76 · 41,023 votes"},{"benchmark":"Long-Horizon-Terminal-Bench · lhtb","score":38.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://zli12321.github.io/LHTB/leaderboard.html","measuredAt":"2026-07-16","scope":"Registered model evaluation · LHTB leaderboard"},{"benchmark":"SWE-bench_Verified · swe bench % resolved","score":80.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","measuredAt":"2026-06-23","scope":"Registered model evaluation · MiniMax-M3 model card"},{"benchmark":"SWE-bench_Pro · SWE Bench Pro","score":59,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","measuredAt":"2026-06-23","scope":"Registered model evaluation · MiniMax-M3 model card"},{"benchmark":"MMMU_Pro · mmmu pro standard 10 options","score":78.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","measuredAt":"2026-06-23","scope":"Registered model evaluation · MiniMax-M3 model card"},{"benchmark":"Video-MME-v2 · video-mme-v2","score":85.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","measuredAt":"2026-06-23","scope":"Registered model evaluation · MiniMax-M3 model card"},{"benchmark":"Claw-Eval · general","score":74.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","measuredAt":"2026-06-23","scope":"Registered model evaluation · MiniMax-M3 model card"},{"benchmark":"apex-agents · apex-agents","score":27.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","measuredAt":"2026-06-23","scope":"Registered model evaluation · MiniMax-M3 model card"},{"benchmark":"Long-Horizon-Terminal-Bench · lhtb solved","score":3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://zli12321.github.io/LHTB/leaderboard.html","measuredAt":"2026-07-16","scope":"Registered model evaluation · LHTB leaderboard"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]},{"id":"hailuo","label":"Hailuo","models":[{"id":"minimax/hailuo-2.3-20260420","name":"Hailuo 2.3","releaseLabel":"2.3","generationId":"2.3","generationLabel":"2.3","variantLabel":"Base","createdAt":"2026-04-20T16:32:20.000Z","kind":"video","contextLength":0,"inputModalities":["text","image"],"outputModalities":["video"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Hailuo 2.3 is a video generation model from MiniMax. It accepts text prompts and reference images as input and generates video output, supporting both text-to-video and image-to-video workflows. It is...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"minimax/hailuo-2.3","url":"https://openrouter.ai/minimax/hailuo-2.3","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Video Arena","url":"https://artificialanalysis.ai/video/leaderboard/text-to-video","note":"Blind preference ranking for video generation"}],"intel":[]},{"id":"minimax/hailuo-03-20260730","name":"H3","releaseLabel":"H3","generationId":"3","generationLabel":"3","variantLabel":"H","createdAt":"2026-07-29T23:10:48.000Z","kind":"video","contextLength":0,"inputModalities":["text","image","video","audio"],"outputModalities":["video"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"MiniMax H3 is a lightweight, open-weights video generation model from MiniMax. It is designed for precise multimodal editing and controlled content generation, including instruction-guided edits, text and brand rendering, and...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"minimax/hailuo-3","url":"https://openrouter.ai/minimax/hailuo-3","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Video Arena","url":"https://artificialanalysis.ai/video/leaderboard/text-to-video","note":"Blind preference ranking for video generation"}],"intel":[]}]},{"id":"speech","label":"Speech","models":[{"id":"minimax/speech-2.8-turbo-20260716","name":"Speech 2.8 Turbo","releaseLabel":"2.8 Turbo","generationId":"2.8","generationLabel":"2.8","variantLabel":"Turbo","createdAt":"2026-07-16T01:06:40.000Z","kind":"audio","contextLength":0,"inputModalities":["text"],"outputModalities":["speech"],"promptPerMillion":60,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"MiniMax Speech 2.8 Turbo is a text-to-speech model from MiniMax. It is suited for applications that generate spoken audio from text and accepts arbitrary MiniMax voice IDs.","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"minimax/speech-2.8-turbo","url":"https://openrouter.ai/minimax/speech-2.8-turbo","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[]},{"id":"minimax/speech-2.8-hd-20260716","name":"Speech 2.8 HD","releaseLabel":"2.8 HD","generationId":"2.8","generationLabel":"2.8","variantLabel":"Hd","createdAt":"2026-07-16T01:06:41.000Z","kind":"audio","contextLength":0,"inputModalities":["text"],"outputModalities":["speech"],"promptPerMillion":100,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"MiniMax Speech 2.8 HD is a text-to-speech model from MiniMax. It is suited for applications that generate spoken audio from text and accepts arbitrary MiniMax voice IDs.","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"minimax/speech-2.8-hd","url":"https://openrouter.ai/minimax/speech-2.8-hd","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[]}]}]},{"id":"bytedance-seed","name":"ByteDance Seed","domain":"seed.bytedance.com","major":false,"modelCount":9,"series":[{"id":"seed","label":"Seed","models":[{"id":"bytedance-seed/seed-1.6-20250625","name":"Seed 1.6","releaseLabel":"1.6","generationId":"1.6","generationLabel":"1.6","variantLabel":"Base","createdAt":"2025-12-23T15:49:57.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["image","text","video"],"outputModalities":["text"],"promptPerMillion":0.25,"completionPerMillion":2,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Seed 1.6 is a general-purpose model released by the ByteDance Seed team. It incorporates multimodal capabilities and adaptive deep thinking with a 256K context window.","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"bytedance-seed/seed-1.6","url":"https://openrouter.ai/bytedance-seed/seed-1.6","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"bytedance-seed/seed-1.6-flash-20250625","name":"Seed 1.6 Flash","releaseLabel":"1.6 Flash","generationId":"1.6","generationLabel":"1.6","variantLabel":"Flash","createdAt":"2025-12-23T15:50:11.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["image","text","video"],"outputModalities":["text"],"promptPerMillion":0.075,"completionPerMillion":0.3,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Seed 1.6 Flash is an ultra-fast multimodal deep thinking model by ByteDance Seed, supporting both text and visual understanding. It features a 256k context window and can generate outputs of...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"bytedance-seed/seed-1.6-flash","url":"https://openrouter.ai/bytedance-seed/seed-1.6-flash","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"bytedance-seed/seed-2.0-mini-20260224","name":"Seed-2.0-Mini","releaseLabel":"2.0-Mini","generationId":"2.0","generationLabel":"2.0","variantLabel":"Mini","createdAt":"2026-02-26T18:38:27.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["text","image","video"],"outputModalities":["text"],"promptPerMillion":0.09999999999999999,"completionPerMillion":0.39999999999999997,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Seed-2.0-mini targets latency-sensitive, high-concurrency, and cost-sensitive scenarios, emphasizing fast response and flexible inference deployment. It delivers performance comparable to ByteDance-Seed-1.6, supports 256k context, four reasoning effort modes (minimal/low/medium/high), multimodal understanding,...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"bytedance-seed/seed-2.0-mini","url":"https://openrouter.ai/bytedance-seed/seed-2.0-mini","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"bytedance-seed/seed-2.0-lite-20260309","name":"Seed-2.0-Lite","releaseLabel":"2.0-Lite","generationId":"2.0","generationLabel":"2.0","variantLabel":"Lite","createdAt":"2026-03-10T15:40:31.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["text","image","video"],"outputModalities":["text"],"promptPerMillion":0.25,"completionPerMillion":2,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Seed-2.0-Lite is a versatile, cost‑efficient enterprise workhorse that delivers strong multimodal and agent capabilities while offering noticeably lower latency, making it a practical default choice for most production workloads across...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"bytedance-seed/seed-2.0-lite","url":"https://openrouter.ai/bytedance-seed/seed-2.0-lite","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"bytedance-seed/seed-2.0-code-20260730","name":"Seed-2.0-Code","releaseLabel":"2.0-Code","generationId":"2.0","generationLabel":"2.0","variantLabel":"Code","createdAt":"2026-08-12T16:05:01.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["text","image","video"],"outputModalities":["text"],"promptPerMillion":0.5,"completionPerMillion":3,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Seed 2.0 Code is a model from ByteDance Seed optimized for agentic coding. It is suited for frontend development, multilingual programming tasks, and coding-agent workflows in tools such as Claude...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"bytedance-seed/seed-2.0-code","url":"https://openrouter.ai/bytedance-seed/seed-2.0-code","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"bytedance-seed/seed-2-1-turbo-20260810","name":"Seed 2.1 Turbo","releaseLabel":"2.1 Turbo","generationId":"2","generationLabel":"2","variantLabel":"1 Turbo","createdAt":"2026-08-12T16:29:36.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["text","image","video"],"outputModalities":["text"],"promptPerMillion":0.5,"completionPerMillion":2.5,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Seed 2.1 Turbo is a multimodal model from ByteDance Seed for coding and long-horizon agent workflows. It is suited for end-to-end software delivery, multi-step task execution, and understanding visual and...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"bytedance-seed/seed-2-1-turbo","url":"https://openrouter.ai/bytedance-seed/seed-2-1-turbo","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]},{"id":"seedream","label":"Seedream","models":[{"id":"bytedance-seed/seedream-4.5-20251203","name":"Seedream 4.5","releaseLabel":"4.5","generationId":"4.5","generationLabel":"4.5","variantLabel":"Base","createdAt":"2025-12-23T19:51:46.000Z","kind":"image","contextLength":4096,"inputModalities":["image","text"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":9.580838323353289,"supportsTools":false,"supportsReasoning":false,"description":"Seedream 4.5 is the latest in-house image generation model developed by ByteDance. Compared with Seedream 4.0, it delivers comprehensive improvements, especially in editing consistency, including better preservation of subject details,...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"bytedance-seed/seedream-4.5","url":"https://openrouter.ai/bytedance-seed/seedream-4.5","free":true,"batch":false}],"benchmarks":[{"benchmark":"AA Image Arena Elo","score":1204,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/image/leaderboard/text-to-image","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[]},{"id":"bytedance-seed/seedream-5-0-pro-20260812","name":"Seedream 5.0 Pro","releaseLabel":"5.0 Pro","generationId":"5","generationLabel":"5","variantLabel":"0 Pro","createdAt":"2026-08-12T23:42:19.000Z","kind":"image","contextLength":0,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":10.7784431137725,"supportsTools":false,"supportsReasoning":false,"description":"Seedream 5.0 Pro is an image generation and editing model from ByteDance Seed. It is suited for commercial visual-production workflows that require precise editing control, lifelike scenes, and natural rendering.","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"bytedance-seed/seedream-5-0-pro","url":"https://openrouter.ai/bytedance-seed/seedream-5-0-pro","free":true,"batch":false}],"benchmarks":[{"benchmark":"AA Image Arena Elo","score":1281,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/image/leaderboard/text-to-image","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[]},{"id":"bytedance-seed/seedream-5-0-lite-20260812","name":"Seedream 5.0 Lite","releaseLabel":"5.0 Lite","generationId":"5","generationLabel":"5","variantLabel":"0 Lite","createdAt":"2026-08-13T19:41:34.000Z","kind":"image","contextLength":0,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":8.38323353293413,"supportsTools":false,"supportsReasoning":false,"description":"Seedream 5.0 Lite is an image generation model from ByteDance Seed. It is suited for professional visual creation that benefits from web-connected retrieval, complex-prompt comprehension, visual references, and broad knowledge...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"bytedance-seed/seedream-5-0-lite","url":"https://openrouter.ai/bytedance-seed/seedream-5-0-lite","free":true,"batch":false}],"benchmarks":[{"benchmark":"AA Image Arena Elo","score":1199,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/image/leaderboard/text-to-image","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[]}]}]},{"id":"microsoft","name":"Microsoft","domain":"microsoft.com","major":false,"modelCount":7,"series":[{"id":"mai-image","label":"MAI Image","models":[{"id":"microsoft/mai-image-2.5","name":"MAI-Image-2.5","releaseLabel":"MAI-Image-2.5","generationId":"2.5","generationLabel":"2.5","variantLabel":"Mai Image","createdAt":"2026-06-02T18:28:16.000Z","kind":"image","contextLength":4096,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":5,"completionPerMillion":0,"imageOutput":47,"supportsTools":false,"supportsReasoning":false,"description":"Microsoft's MAI-Image-2.5 is a high-quality image generation model available via Azure AI Foundry. It produces photorealistic and artistic images from text prompts with support for various aspect ratios.","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"microsoft/mai-image-2.5","url":"https://openrouter.ai/microsoft/mai-image-2.5","free":false,"batch":false}],"benchmarks":[{"benchmark":"AA Image Arena Elo","score":1304,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/image/leaderboard/text-to-image","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[{"id":"6dbbb29c-5288-4ab3-b491-0e9aa3e3ad90","title":"promptbase","href":"/intel/6dbbb29c-5288-4ab3-b491-0e9aa3e3ad90","hook":"Microsoft's distilled prompt-engineering techniques and benchmarks from running LLMs at scale. Less \"10 magic prompts,\" more the methods that actually hold u…"},{"id":"c06033e3-1f57-4207-9545-b14b3bd9f32c","title":"Satya Nadella — Microsoft’s AGI plan & quantum breakthrough","href":"/intel/c06033e3-1f57-4207-9545-b14b3bd9f32c","hook":"Nadella lays out a concrete thesis on why intelligence pricing keeps falling and why AI won't be winner-take-all, plus early signals on gaming world models (…"},{"id":"71615908-05f9-4e45-b58b-32f4dfc9522d","title":"Satya Nadella — How Microsoft is preparing for AGI","href":"/intel/71615908-05f9-4e45-b58b-32f4dfc9522d","hook":"Microsoft's CEO on the datacenter buildout behind frontier AI — CAPEX, GB300 clusters and how the platform bets translate to developer tooling."}]},{"id":"microsoft/mai-image-2.5-pro-20260723","name":"MAI-Image-2.5 Pro","releaseLabel":"MAI-Image-2.5 Pro","generationId":"2.5","generationLabel":"2.5","variantLabel":"Pro","createdAt":"2026-07-23T17:28:21.000Z","kind":"image","contextLength":4096,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":5,"completionPerMillion":0,"imageOutput":108,"supportsTools":false,"supportsReasoning":false,"description":"Microsoft's MAI-Image-2.5 is a high-quality image generation model available via Azure AI Foundry. It produces photorealistic and artistic images from text prompts with support for various aspect ratios.","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"microsoft/mai-image-2.5-pro","url":"https://openrouter.ai/microsoft/mai-image-2.5-pro","free":false,"batch":false}],"benchmarks":[{"benchmark":"AA Image Arena Elo","score":1294,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/image/leaderboard/text-to-image","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[{"id":"6dbbb29c-5288-4ab3-b491-0e9aa3e3ad90","title":"promptbase","href":"/intel/6dbbb29c-5288-4ab3-b491-0e9aa3e3ad90","hook":"Microsoft's distilled prompt-engineering techniques and benchmarks from running LLMs at scale. Less \"10 magic prompts,\" more the methods that actually hold u…"},{"id":"c06033e3-1f57-4207-9545-b14b3bd9f32c","title":"Satya Nadella — Microsoft’s AGI plan & quantum breakthrough","href":"/intel/c06033e3-1f57-4207-9545-b14b3bd9f32c","hook":"Nadella lays out a concrete thesis on why intelligence pricing keeps falling and why AI won't be winner-take-all, plus early signals on gaming world models (…"},{"id":"71615908-05f9-4e45-b58b-32f4dfc9522d","title":"Satya Nadella — How Microsoft is preparing for AGI","href":"/intel/71615908-05f9-4e45-b58b-32f4dfc9522d","hook":"Microsoft's CEO on the datacenter buildout behind frontier AI — CAPEX, GB300 clusters and how the platform bets translate to developer tooling."}]}]},{"id":"mai-voice","label":"MAI Voice","models":[{"id":"microsoft/mai-voice-2","name":"MAI-Voice-2","releaseLabel":"MAI-Voice-2","generationId":"2","generationLabel":"2","variantLabel":"Mai Voice","createdAt":"2026-06-02T18:31:37.000Z","kind":"audio","contextLength":0,"inputModalities":["text"],"outputModalities":["speech"],"promptPerMillion":22,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"MAI-Voice-2 is an expressive text-to-speech model from Microsoft. It is suited for conversational assistants, media narration, accessibility, education, and other long-form voice applications. It supports 15 languages across 18 locales,...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"microsoft/mai-voice-2","url":"https://openrouter.ai/microsoft/mai-voice-2","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[{"id":"6dbbb29c-5288-4ab3-b491-0e9aa3e3ad90","title":"promptbase","href":"/intel/6dbbb29c-5288-4ab3-b491-0e9aa3e3ad90","hook":"Microsoft's distilled prompt-engineering techniques and benchmarks from running LLMs at scale. Less \"10 magic prompts,\" more the methods that actually hold u…"},{"id":"c06033e3-1f57-4207-9545-b14b3bd9f32c","title":"Satya Nadella — Microsoft’s AGI plan & quantum breakthrough","href":"/intel/c06033e3-1f57-4207-9545-b14b3bd9f32c","hook":"Nadella lays out a concrete thesis on why intelligence pricing keeps falling and why AI won't be winner-take-all, plus early signals on gaming world models (…"},{"id":"71615908-05f9-4e45-b58b-32f4dfc9522d","title":"Satya Nadella — How Microsoft is preparing for AGI","href":"/intel/71615908-05f9-4e45-b58b-32f4dfc9522d","hook":"Microsoft's CEO on the datacenter buildout behind frontier AI — CAPEX, GB300 clusters and how the platform bets translate to developer tooling."}]},{"id":"microsoft/mai-voice-2-flash-20260723","name":"MAI-Voice-2-Flash","releaseLabel":"MAI-Voice-2-Flash","generationId":"2","generationLabel":"2","variantLabel":"Flash","createdAt":"2026-07-23T15:54:40.000Z","kind":"audio","contextLength":0,"inputModalities":["text"],"outputModalities":["speech"],"promptPerMillion":15,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"MAI-Voice-2-Flash is a low-latency text-to-speech model from Microsoft for voice agents, assistants, call centers, accessibility, narration, and other interactive applications. It generates expressive 24 kHz mono speech across 15 languages...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"microsoft/mai-voice-2-flash","url":"https://openrouter.ai/microsoft/mai-voice-2-flash","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[{"id":"6dbbb29c-5288-4ab3-b491-0e9aa3e3ad90","title":"promptbase","href":"/intel/6dbbb29c-5288-4ab3-b491-0e9aa3e3ad90","hook":"Microsoft's distilled prompt-engineering techniques and benchmarks from running LLMs at scale. Less \"10 magic prompts,\" more the methods that actually hold u…"},{"id":"c06033e3-1f57-4207-9545-b14b3bd9f32c","title":"Satya Nadella — Microsoft’s AGI plan & quantum breakthrough","href":"/intel/c06033e3-1f57-4207-9545-b14b3bd9f32c","hook":"Nadella lays out a concrete thesis on why intelligence pricing keeps falling and why AI won't be winner-take-all, plus early signals on gaming world models (…"},{"id":"71615908-05f9-4e45-b58b-32f4dfc9522d","title":"Satya Nadella — How Microsoft is preparing for AGI","href":"/intel/71615908-05f9-4e45-b58b-32f4dfc9522d","hook":"Microsoft's CEO on the datacenter buildout behind frontier AI — CAPEX, GB300 clusters and how the platform bets translate to developer tooling."}]}]},{"id":"mai-transcribe","label":"MAI Transcribe","models":[{"id":"microsoft/mai-transcribe-1.5","name":"MAI-Transcribe 1.5","releaseLabel":"MAI-Transcribe 1.5","generationId":"1.5","generationLabel":"1.5","variantLabel":"Mai Transcribe","createdAt":"2026-06-02T18:31:35.000Z","kind":"audio","contextLength":0,"inputModalities":["audio"],"outputModalities":["transcription"],"promptPerMillion":360000,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"MAI-Transcribe 1.5 is a multilingual speech-to-text model from Microsoft AI. It is suited for captions, call transcription, subtitling, accessibility, and other voice-enabled applications, with reliable transcription across 43 languages, diverse...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"microsoft/mai-transcribe-1.5","url":"https://openrouter.ai/microsoft/mai-transcribe-1.5","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[{"id":"6dbbb29c-5288-4ab3-b491-0e9aa3e3ad90","title":"promptbase","href":"/intel/6dbbb29c-5288-4ab3-b491-0e9aa3e3ad90","hook":"Microsoft's distilled prompt-engineering techniques and benchmarks from running LLMs at scale. Less \"10 magic prompts,\" more the methods that actually hold u…"},{"id":"c06033e3-1f57-4207-9545-b14b3bd9f32c","title":"Satya Nadella — Microsoft’s AGI plan & quantum breakthrough","href":"/intel/c06033e3-1f57-4207-9545-b14b3bd9f32c","hook":"Nadella lays out a concrete thesis on why intelligence pricing keeps falling and why AI won't be winner-take-all, plus early signals on gaming world models (…"},{"id":"71615908-05f9-4e45-b58b-32f4dfc9522d","title":"Satya Nadella — How Microsoft is preparing for AGI","href":"/intel/71615908-05f9-4e45-b58b-32f4dfc9522d","hook":"Microsoft's CEO on the datacenter buildout behind frontier AI — CAPEX, GB300 clusters and how the platform bets translate to developer tooling."}]}]},{"id":"phi","label":"PHI","models":[{"id":"microsoft/phi-4","name":"Phi 4","releaseLabel":"4","generationId":"4","generationLabel":"4","variantLabel":"Base","createdAt":"2025-01-10T06:17:52.000Z","kind":"language","contextLength":16384,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.07,"completionPerMillion":0.14,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"[Microsoft Research](/microsoft) Phi-4 is designed to perform well in complex reasoning tasks and can operate efficiently in situations with limited memory or where quick responses are needed. At 14 billion...","huggingFaceId":"microsoft/phi-4","routes":[{"provider":"OpenRouter","modelId":"microsoft/phi-4","url":"https://openrouter.ai/microsoft/phi-4","free":false,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":42.3,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2025_04_02.csv","measuredAt":"2025-04-02","scope":"Mean across that public release's task columns"},{"benchmark":"AA Intelligence Index","score":4.6,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/phi-4","measuredAt":"2026-08-27","scope":"Current public index; highest tested effort configuration"},{"benchmark":"SciCode","score":26,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/phi-4","measuredAt":"2026-08-27","scope":"Scientific coding evaluation"},{"benchmark":"AA Omniscience Index","score":-55.7,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/phi-4","measuredAt":"2026-08-27","scope":"Current public knowledge and hallucination index"},{"benchmark":"MLCR overall","score":18.3,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/phi-4","measuredAt":"2026-08-27","scope":"Machine-learning code reasoning evaluation"},{"benchmark":"GDPval-AA normalized","score":42,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/phi-4","measuredAt":"2026-08-27","scope":"Normalized real-world knowledge-work evaluation"},{"benchmark":"τ²-bench","score":0,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/phi-4","measuredAt":"2026-08-27","scope":"Tool-agent benchmark across supported domains"},{"benchmark":"Terminal-Bench Hard","score":3.8,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/phi-4","measuredAt":"2026-08-27","scope":"Hard terminal-agent tasks"},{"benchmark":"LCR","score":0,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/phi-4","measuredAt":"2026-08-27","scope":"Long-context reasoning evaluation"},{"benchmark":"IFBench","score":23.5,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/phi-4","measuredAt":"2026-08-27","scope":"Instruction-following evaluation"},{"benchmark":"Humanity's Last Exam","score":3.8,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/phi-4","measuredAt":"2026-08-27","scope":"Broad expert-level knowledge evaluation"},{"benchmark":"GPQA","score":57.5,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/phi-4","measuredAt":"2026-08-27","scope":"Graduate-level science questions"},{"benchmark":"CritPt","score":0,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/phi-4","measuredAt":"2026-08-27","scope":"Critical-point reasoning evaluation"},{"benchmark":"LiveCodeBench","score":23.1,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/phi-4","measuredAt":"2026-08-27","scope":"Contamination-resistant coding tasks"},{"benchmark":"AIME 2025","score":18,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/phi-4","measuredAt":"2026-08-27","scope":"Competition mathematics evaluation"},{"benchmark":"LMArena overall","score":1217,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 292 · 24,126 votes"},{"benchmark":"MMLU-Pro · mmlu pro","score":70.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/c2/37/c23724de-5d83-4794-abcb-9af3ce249886.json","measuredAt":"2026-06-30","scope":"Registered model evaluation · EvalEval"},{"benchmark":"LEXam · open question","score":38.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://lexam-benchmark.github.io/","measuredAt":"2026-06-02","scope":"Registered model evaluation · LEXam Leaderboard"},{"benchmark":"LEXam · mcq 4 choices","score":40.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://lexam-benchmark.github.io/","measuredAt":"2026-06-02","scope":"Registered model evaluation · LEXam Leaderboard"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"c67698d3-7a61-484b-84e5-f8eae18b14d7","title":"How to really stop your agents from making the same mistakes","href":"/intel/c67698d3-7a61-484b-84e5-f8eae18b14d7","hook":"A skeptical look at agent evaluation and observability — using LangSmith's sophistication as a foil — and what actually stops an agent repeating failures ver…"},{"id":"3b75707e-9388-437e-97b1-4d7ea7cc3a9c","title":"AI Memory Management Philosophy","href":"/intel/3b75707e-9388-437e-97b1-4d7ea7cc3a9c","hook":"Argues that agents should store learned lessons as pull requests to shared, code-reviewed skill files instead of private memory — private memory accumulates …"},{"id":"6dbbb29c-5288-4ab3-b491-0e9aa3e3ad90","title":"promptbase","href":"/intel/6dbbb29c-5288-4ab3-b491-0e9aa3e3ad90","hook":"Microsoft's distilled prompt-engineering techniques and benchmarks from running LLMs at scale. Less \"10 magic prompts,\" more the methods that actually hold u…"}]}]},{"id":"wizardlm","label":"Wizardlm","models":[{"id":"microsoft/wizardlm-2-8x22b","name":"WizardLM-2 8x22B","releaseLabel":"2 8x22B","generationId":"2","generationLabel":"2","variantLabel":"8x22b","createdAt":"2024-04-16T00:00:00.000Z","kind":"language","contextLength":65535,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.62,"completionPerMillion":0.62,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"WizardLM-2 8x22B is Microsoft AI's most advanced Wizard model. It demonstrates highly competitive performance compared to leading proprietary models, and it consistently outperforms all existing state-of-the-art opensource models. It is...","huggingFaceId":"microsoft/WizardLM-2-8x22B","routes":[{"provider":"OpenRouter","modelId":"microsoft/wizardlm-2-8x22b","url":"https://openrouter.ai/microsoft/wizardlm-2-8x22b","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"6dbbb29c-5288-4ab3-b491-0e9aa3e3ad90","title":"promptbase","href":"/intel/6dbbb29c-5288-4ab3-b491-0e9aa3e3ad90","hook":"Microsoft's distilled prompt-engineering techniques and benchmarks from running LLMs at scale. Less \"10 magic prompts,\" more the methods that actually hold u…"},{"id":"c06033e3-1f57-4207-9545-b14b3bd9f32c","title":"Satya Nadella — Microsoft’s AGI plan & quantum breakthrough","href":"/intel/c06033e3-1f57-4207-9545-b14b3bd9f32c","hook":"Nadella lays out a concrete thesis on why intelligence pricing keeps falling and why AI won't be winner-take-all, plus early signals on gaming world models (…"},{"id":"71615908-05f9-4e45-b58b-32f4dfc9522d","title":"Satya Nadella — How Microsoft is preparing for AGI","href":"/intel/71615908-05f9-4e45-b58b-32f4dfc9522d","hook":"Microsoft's CEO on the datacenter buildout behind frontier AI — CAPEX, GB300 clusters and how the platform bets translate to developer tooling."}]}]}]},{"id":"perplexity","name":"Perplexity","domain":"perplexity.ai","major":false,"modelCount":7,"series":[{"id":"sonar","label":"Sonar","models":[{"id":"perplexity/sonar","name":"Sonar","releaseLabel":"Sonar","generationId":"releases","generationLabel":"Releases","variantLabel":"Sonar","createdAt":"2025-01-27T21:36:48.000Z","kind":"multimodal","contextLength":127072,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":1,"completionPerMillion":1,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Sonar is lightweight, affordable, fast, and simple to use — now featuring citations and the ability to customize sources. It is designed for companies seeking to integrate lightweight question-and-answer features...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"perplexity/sonar","url":"https://openrouter.ai/perplexity/sonar","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"4e5f6060-b7b5-4f90-9062-a21635833125","title":"The Harness Is Everything: What Cursor, Claude Code, and Perplexity Actually Built","href":"/intel/4e5f6060-b7b5-4f90-9062-a21635833125","hook":"The thesis that you are not picking the wrong model, you are building the wrong environment around it — with Cursor, Claude Code, and Perplexity as evidence.…"}]},{"id":"perplexity/sonar-deep-research","name":"Sonar Deep Research","releaseLabel":"Deep Research","generationId":"releases","generationLabel":"Releases","variantLabel":"Deep Research","createdAt":"2025-03-07T01:34:06.000Z","kind":"language","contextLength":128000,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":2,"completionPerMillion":8,"imageOutput":null,"supportsTools":false,"supportsReasoning":true,"description":"Sonar Deep Research is a research-focused model designed for multi-step retrieval, synthesis, and reasoning across complex topics. It autonomously searches, reads, and evaluates sources, refining its approach as it gathers...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"perplexity/sonar-deep-research","url":"https://openrouter.ai/perplexity/sonar-deep-research","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"4e5f6060-b7b5-4f90-9062-a21635833125","title":"The Harness Is Everything: What Cursor, Claude Code, and Perplexity Actually Built","href":"/intel/4e5f6060-b7b5-4f90-9062-a21635833125","hook":"The thesis that you are not picking the wrong model, you are building the wrong environment around it — with Cursor, Claude Code, and Perplexity as evidence.…"}]},{"id":"perplexity/sonar-pro","name":"Sonar Pro","releaseLabel":"Pro","generationId":"releases","generationLabel":"Releases","variantLabel":"Pro","createdAt":"2025-03-07T01:53:43.000Z","kind":"multimodal","contextLength":200000,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":3,"completionPerMillion":15,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) For enterprises seeking more advanced capabilities, the Sonar Pro API can handle in-depth, multi-step queries with added extensibility, like...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"perplexity/sonar-pro","url":"https://openrouter.ai/perplexity/sonar-pro","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"4e5f6060-b7b5-4f90-9062-a21635833125","title":"The Harness Is Everything: What Cursor, Claude Code, and Perplexity Actually Built","href":"/intel/4e5f6060-b7b5-4f90-9062-a21635833125","hook":"The thesis that you are not picking the wrong model, you are building the wrong environment around it — with Cursor, Claude Code, and Perplexity as evidence.…"}]},{"id":"perplexity/sonar-reasoning-pro","name":"Sonar Reasoning Pro","releaseLabel":"Reasoning Pro","generationId":"releases","generationLabel":"Releases","variantLabel":"Reasoning Pro","createdAt":"2025-03-07T02:08:28.000Z","kind":"multimodal","contextLength":128000,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":2,"completionPerMillion":8,"imageOutput":null,"supportsTools":false,"supportsReasoning":true,"description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) Sonar Reasoning Pro is a premier reasoning model powered by DeepSeek R1 with Chain of Thought (CoT). Designed for...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"perplexity/sonar-reasoning-pro","url":"https://openrouter.ai/perplexity/sonar-reasoning-pro","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"4e5f6060-b7b5-4f90-9062-a21635833125","title":"The Harness Is Everything: What Cursor, Claude Code, and Perplexity Actually Built","href":"/intel/4e5f6060-b7b5-4f90-9062-a21635833125","hook":"The thesis that you are not picking the wrong model, you are building the wrong environment around it — with Cursor, Claude Code, and Perplexity as evidence.…"}]},{"id":"perplexity/sonar-pro-search","name":"Sonar Pro Search","releaseLabel":"Pro Search","generationId":"releases","generationLabel":"Releases","variantLabel":"Pro Search","createdAt":"2025-10-30T19:59:26.000Z","kind":"multimodal","contextLength":200000,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":3,"completionPerMillion":15,"imageOutput":null,"supportsTools":false,"supportsReasoning":true,"description":"Exclusively available on the OpenRouter API, Sonar Pro's new Pro Search mode is Perplexity's most advanced agentic search system. It is designed for deeper reasoning and analysis. Pricing is based...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"perplexity/sonar-pro-search","url":"https://openrouter.ai/perplexity/sonar-pro-search","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"4e5f6060-b7b5-4f90-9062-a21635833125","title":"The Harness Is Everything: What Cursor, Claude Code, and Perplexity Actually Built","href":"/intel/4e5f6060-b7b5-4f90-9062-a21635833125","hook":"The thesis that you are not picking the wrong model, you are building the wrong environment around it — with Cursor, Claude Code, and Perplexity as evidence.…"}]}]},{"id":"pplx","label":"Pplx","models":[{"id":"perplexity/pplx-embed-v1-0.6B","name":"Embed V1 0.6B","releaseLabel":"Embed V1 0.6B","generationId":"v1","generationLabel":"V1","variantLabel":"0.6B","createdAt":"2026-03-16T01:34:28.000Z","kind":"embedding","contextLength":32000,"inputModalities":["text"],"outputModalities":["embeddings"],"promptPerMillion":0.004,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"pplx-embed-v1-0.6B is one of Perplexity's state-of-the-art text embedding models built for real-world, web-scale retrieval. pplx-embed-v1 is optimized for standard dense text retrieval with the 0.6B parameter model targeting lightweight, low-latency...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"perplexity/pplx-embed-v1-0.6b","url":"https://openrouter.ai/perplexity/pplx-embed-v1-0.6b","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[{"id":"4e5f6060-b7b5-4f90-9062-a21635833125","title":"The Harness Is Everything: What Cursor, Claude Code, and Perplexity Actually Built","href":"/intel/4e5f6060-b7b5-4f90-9062-a21635833125","hook":"The thesis that you are not picking the wrong model, you are building the wrong environment around it — with Cursor, Claude Code, and Perplexity as evidence.…"}]},{"id":"perplexity/pplx-embed-v1-4B","name":"Embed V1 4B","releaseLabel":"Embed V1 4B","generationId":"v1","generationLabel":"V1","variantLabel":"4B","createdAt":"2026-03-16T01:42:52.000Z","kind":"embedding","contextLength":32000,"inputModalities":["text"],"outputModalities":["embeddings"],"promptPerMillion":0.03,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"pplx-embed-v1 -4B is one of Perplexity's state-of-the-art text embedding models built for real-world, web-scale retrieval. pplx-embed-v1 is optimized for standard dense text retrieval with the 4B parameter model maximizing retrieval...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"perplexity/pplx-embed-v1-4b","url":"https://openrouter.ai/perplexity/pplx-embed-v1-4b","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[{"id":"4e5f6060-b7b5-4f90-9062-a21635833125","title":"The Harness Is Everything: What Cursor, Claude Code, and Perplexity Actually Built","href":"/intel/4e5f6060-b7b5-4f90-9062-a21635833125","hook":"The thesis that you are not picking the wrong model, you are building the wrong environment around it — with Cursor, Claude Code, and Perplexity as evidence.…"}]}]}]},{"id":"voyageai","name":"Voyage AI","domain":"voyageai.com","major":false,"modelCount":7,"series":[{"id":"voyage","label":"Voyage","models":[{"id":"voyageai/voyage-4-large-20260727","name":"voyage-4-large","releaseLabel":"4-large","generationId":"4","generationLabel":"4","variantLabel":"Large","createdAt":"2026-07-27T21:43:44.000Z","kind":"embedding","contextLength":32000,"inputModalities":["text"],"outputModalities":["embeddings"],"promptPerMillion":0.12,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"voyage-4-large is a state-of-the-art general-purpose and multilingual embedding optimized for retrieval quality. Enabled by Matryoshka learning and quantization-aware training, voyage-4-large supports embeddings in 2048, 1024, 512, and 256 dimensions, with...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"voyageai/voyage-4-large","url":"https://openrouter.ai/voyageai/voyage-4-large","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[]},{"id":"voyageai/voyage-4-20260727","name":"voyage-4","releaseLabel":"4","generationId":"4","generationLabel":"4","variantLabel":"Base","createdAt":"2026-07-27T21:43:46.000Z","kind":"embedding","contextLength":32000,"inputModalities":["text"],"outputModalities":["embeddings"],"promptPerMillion":0.06,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"voyage-4 is a general-purpose (including multilingual) embedding model optimized for retrieval/search and AI applications. voyage-4 supports embeddings in 2048, 1024, 512, and 256 dimensions, with multiple quantization options. Learn more...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"voyageai/voyage-4","url":"https://openrouter.ai/voyageai/voyage-4","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[]},{"id":"voyageai/voyage-4-lite-20260727","name":"voyage-4-lite","releaseLabel":"4-lite","generationId":"4","generationLabel":"4","variantLabel":"Lite","createdAt":"2026-07-27T21:43:47.000Z","kind":"embedding","contextLength":32000,"inputModalities":["text"],"outputModalities":["embeddings"],"promptPerMillion":0.02,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"voyage-4-lite is a lightweight, general-purpose embedding model optimized for low latency and cost. Enabled by Matryoshka learning and quantization-aware training, voyage-4-lite supports embeddings in 2048, 1024, 512, and 256 dimensions,...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"voyageai/voyage-4-lite","url":"https://openrouter.ai/voyageai/voyage-4-lite","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[]},{"id":"voyageai/voyage-multimodal-3.5-20260727","name":"voyage-multimodal-3.5","releaseLabel":"multimodal-3.5","generationId":"3.5","generationLabel":"3.5","variantLabel":"Multimodal","createdAt":"2026-07-27T21:43:49.000Z","kind":"embedding","contextLength":32000,"inputModalities":["text","image"],"outputModalities":["embeddings"],"promptPerMillion":0.12,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"voyage-multimodal-3.5 is a state-of-the-art multimodal embedding model capable of vectorizing not only text, images, and video individually, but also content that interleaves all three modalities. It delivers excellent performance for...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"voyageai/voyage-multimodal-3.5","url":"https://openrouter.ai/voyageai/voyage-multimodal-3.5","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[]},{"id":"voyageai/voyage-code-4-20260812","name":"voyage-code-4","releaseLabel":"code-4","generationId":"4","generationLabel":"4","variantLabel":"Code","createdAt":"2026-08-13T16:01:52.000Z","kind":"embedding","contextLength":32000,"inputModalities":["text"],"outputModalities":["embeddings"],"promptPerMillion":0.12,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"voyage-code-4 is a code embedding model from Voyage AI, a MongoDB company. It is designed for coding agents and code retrieval, with Matryoshka embeddings at 2048, 1024, 512, and 256...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"voyageai/voyage-code-4","url":"https://openrouter.ai/voyageai/voyage-code-4","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[]}]},{"id":"rerank","label":"Rerank","models":[{"id":"voyageai/rerank-2.5-20260727","name":"rerank-2.5","releaseLabel":"2.5","generationId":"2.5","generationLabel":"2.5","variantLabel":"Base","createdAt":"2026-07-27T21:43:50.000Z","kind":"language","contextLength":32000,"inputModalities":["text"],"outputModalities":["rerank"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"rerank-2.5 is a cutting-edge reranker optimized for quality, delivering a 7.94% improvement in retrieval accuracy over Cohere Rerank v3.5 across 93 datasets. It also outperformed Cohere Rerank v3.5 by 12.70%...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"voyageai/rerank-2.5","url":"https://openrouter.ai/voyageai/rerank-2.5","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"voyageai/rerank-2.5-lite-20260727","name":"rerank-2.5-lite","releaseLabel":"2.5-lite","generationId":"2.5","generationLabel":"2.5","variantLabel":"Lite","createdAt":"2026-07-27T21:43:51.000Z","kind":"language","contextLength":32000,"inputModalities":["text"],"outputModalities":["rerank"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"rerank-2.5-lite is a reranker optimized for both latency and quality, delivering a 7.16% improvement in retrieval accuracy over Cohere Rerank v3.5 across 93 datasets. It also outperformed Cohere Rerank v3.5...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"voyageai/rerank-2.5-lite","url":"https://openrouter.ai/voyageai/rerank-2.5-lite","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]}]},{"id":"alibaba","name":"Alibaba","domain":"alibabacloud.com","major":false,"modelCount":6,"series":[{"id":"wan","label":"WAN","models":[{"id":"alibaba/wan-2.6-20260327","name":"Wan 2.6","releaseLabel":"2.6","generationId":"2.6","generationLabel":"2.6","variantLabel":"Base","createdAt":"2026-03-28T00:53:10.000Z","kind":"video","contextLength":0,"inputModalities":["text","image"],"outputModalities":["video"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Alibaba's most advanced video generation model, supporting over 10 visual creation capabilities in a unified system. Wan 2.6 generates 1080p video at 24fps from text, images, reference videos, or audio,...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"alibaba/wan-2.6","url":"https://openrouter.ai/alibaba/wan-2.6","free":true,"batch":false}],"benchmarks":[{"benchmark":"AA Video Arena Elo","score":1026,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/video/leaderboard/text-to-video","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Video Arena","url":"https://artificialanalysis.ai/video/leaderboard/text-to-video","note":"Blind preference ranking for video generation"}],"intel":[{"id":"39eb0f39-568d-46dd-8ef8-3b716bd121d2","title":"Deep Agents Evaluation Framework · article","href":"/intel/39eb0f39-568d-46dd-8ef8-3b716bd121d2","hook":"A systematic framework for curating eval data and measuring agent behavior, instead of eyeballing whether it \"seems better.\" Read it if you want your agent i…"},{"id":"6c6ecc81-ac5c-40c7-aca6-ae944c024a42","title":"Introducing EmDash — the spiritual successor to WordPress that solves plugin security","href":"/intel/6c6ecc81-ac5c-40c7-aca6-ae944c024a42","hook":"Cloudflare's pitch for EmDash, a serverless WordPress successor that kills the plugin-security problem. Worth reading if you build content sites and want to …"},{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"}]},{"id":"alibaba/wan-2.7-20260414","name":"Wan 2.7","releaseLabel":"2.7","generationId":"2.7","generationLabel":"2.7","variantLabel":"Base","createdAt":"2026-04-15T00:02:42.000Z","kind":"video","contextLength":0,"inputModalities":["text","image"],"outputModalities":["video"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Wan 2.7 is a video generation model from Alibaba. It supports text-to-video, image-to-video with first and last frame control, and reference-to-video, where multiple reference images guide the style and content...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"alibaba/wan-2.7","url":"https://openrouter.ai/alibaba/wan-2.7","free":true,"batch":false}],"benchmarks":[{"benchmark":"AA Video Arena Elo","score":1106,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/video/leaderboard/text-to-video","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Video Arena","url":"https://artificialanalysis.ai/video/leaderboard/text-to-video","note":"Blind preference ranking for video generation"}],"intel":[{"id":"39eb0f39-568d-46dd-8ef8-3b716bd121d2","title":"Deep Agents Evaluation Framework · article","href":"/intel/39eb0f39-568d-46dd-8ef8-3b716bd121d2","hook":"A systematic framework for curating eval data and measuring agent behavior, instead of eyeballing whether it \"seems better.\" Read it if you want your agent i…"},{"id":"6c6ecc81-ac5c-40c7-aca6-ae944c024a42","title":"Introducing EmDash — the spiritual successor to WordPress that solves plugin security","href":"/intel/6c6ecc81-ac5c-40c7-aca6-ae944c024a42","hook":"Cloudflare's pitch for EmDash, a serverless WordPress successor that kills the plugin-security problem. Worth reading if you build content sites and want to …"},{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"}]},{"id":"alibaba/wan-3.0-20260824","name":"Wan 3.0","releaseLabel":"3.0","generationId":"3.0","generationLabel":"3.0","variantLabel":"Base","createdAt":"2026-08-24T20:37:36.000Z","kind":"video","contextLength":0,"inputModalities":["text","image"],"outputModalities":["video"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Wan 3.0 is a video generation model from Alibaba for text-to-video, image-to-video, and reference-guided video generation. It produces 480p, 720p, or 1080p video with durations from 2 to 30 seconds.","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"alibaba/wan-3.0","url":"https://openrouter.ai/alibaba/wan-3.0","free":true,"batch":false}],"benchmarks":[{"benchmark":"AA Video Arena Elo","score":1240,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/video/leaderboard/text-to-video","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Video Arena","url":"https://artificialanalysis.ai/video/leaderboard/text-to-video","note":"Blind preference ranking for video generation"}],"intel":[{"id":"39eb0f39-568d-46dd-8ef8-3b716bd121d2","title":"Deep Agents Evaluation Framework · article","href":"/intel/39eb0f39-568d-46dd-8ef8-3b716bd121d2","hook":"A systematic framework for curating eval data and measuring agent behavior, instead of eyeballing whether it \"seems better.\" Read it if you want your agent i…"},{"id":"6c6ecc81-ac5c-40c7-aca6-ae944c024a42","title":"Introducing EmDash — the spiritual successor to WordPress that solves plugin security","href":"/intel/6c6ecc81-ac5c-40c7-aca6-ae944c024a42","hook":"Cloudflare's pitch for EmDash, a serverless WordPress successor that kills the plugin-security problem. Worth reading if you build content sites and want to …"},{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"}]},{"id":"alibaba/wan-3.0-prime-20260827","name":"Wan 3.0 Prime","releaseLabel":"3.0 Prime","generationId":"3.0","generationLabel":"3.0","variantLabel":"Prime","createdAt":"2026-08-27T20:50:00.000Z","kind":"video","contextLength":0,"inputModalities":["text","image"],"outputModalities":["video"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Wan 3.0 Prime is a fast-mode variant of [Wan 3.0](https://openrouter.ai/alibaba/wan-3.0) from Alibaba. It supports text-to-video and first-frame image-to-video generation.","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"alibaba/wan-3.0-prime","url":"https://openrouter.ai/alibaba/wan-3.0-prime","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Video Arena","url":"https://artificialanalysis.ai/video/leaderboard/text-to-video","note":"Blind preference ranking for video generation"}],"intel":[{"id":"39eb0f39-568d-46dd-8ef8-3b716bd121d2","title":"Deep Agents Evaluation Framework · article","href":"/intel/39eb0f39-568d-46dd-8ef8-3b716bd121d2","hook":"A systematic framework for curating eval data and measuring agent behavior, instead of eyeballing whether it \"seems better.\" Read it if you want your agent i…"},{"id":"6c6ecc81-ac5c-40c7-aca6-ae944c024a42","title":"Introducing EmDash — the spiritual successor to WordPress that solves plugin security","href":"/intel/6c6ecc81-ac5c-40c7-aca6-ae944c024a42","hook":"Cloudflare's pitch for EmDash, a serverless WordPress successor that kills the plugin-security problem. Worth reading if you build content sites and want to …"},{"id":"31d65fb4-ce2b-45f0-a92b-f90247cc9039","title":"OpenAI Model Spec","href":"/intel/31d65fb4-ce2b-45f0-a92b-f90247cc9039","hook":"The actual document defining how OpenAI wants its models to behave — the priorities behind the refusals and tone you fight with daily. Read it to understand …"}]}]},{"id":"happyhorse","label":"Happyhorse","models":[{"id":"alibaba/happyhorse-1.0-20260624","name":"HappyHorse 1.0","releaseLabel":"1.0","generationId":"1.0","generationLabel":"1.0","variantLabel":"Base","createdAt":"2026-06-24T00:18:44.000Z","kind":"video","contextLength":0,"inputModalities":["text","image"],"outputModalities":["video"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"HappyHorse 1.0 is a video generation model from Alibaba. It generates short videos from a text prompt, a single starting image, or a set of reference images, with output up...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"alibaba/happyhorse-1.0","url":"https://openrouter.ai/alibaba/happyhorse-1.0","free":true,"batch":false}],"benchmarks":[{"benchmark":"AA Video Arena Elo","score":1121,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/video/leaderboard/text-to-video","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Video Arena","url":"https://artificialanalysis.ai/video/leaderboard/text-to-video","note":"Blind preference ranking for video generation"}],"intel":[]},{"id":"alibaba/happyhorse-1.1-20260624","name":"HappyHorse 1.1","releaseLabel":"1.1","generationId":"1.1","generationLabel":"1.1","variantLabel":"Base","createdAt":"2026-06-24T02:54:03.000Z","kind":"video","contextLength":0,"inputModalities":["text","image"],"outputModalities":["video"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"HappyHorse 1.1 is a video generation model from Alibaba. It generates short videos from a text prompt, a single starting image, or a set of reference images, with output up...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"alibaba/happyhorse-1.1","url":"https://openrouter.ai/alibaba/happyhorse-1.1","free":true,"batch":false}],"benchmarks":[{"benchmark":"AA Video Arena Elo","score":1145,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/video/leaderboard/text-to-video","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Video Arena","url":"https://artificialanalysis.ai/video/leaderboard/text-to-video","note":"Blind preference ranking for video generation"}],"intel":[]}]}]},{"id":"black-forest-labs","name":"Black Forest Labs","domain":"bfl.ai","major":false,"modelCount":6,"series":[{"id":"flux","label":"FLUX","models":[{"id":"black-forest-labs/flux.2-pro","name":"FLUX.2 Pro","releaseLabel":".2 Pro","generationId":"releases","generationLabel":"Releases","variantLabel":".2 Pro","createdAt":"2025-11-25T00:24:34.000Z","kind":"image","contextLength":46864,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":7.32421875,"supportsTools":false,"supportsReasoning":false,"description":"A high-end image generation and editing model focused on frontier-level visual quality and reliability. It delivers strong prompt adherence, stable lighting, sharp textures, and consistent character/style reproduction across multi-reference inputs....","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"black-forest-labs/flux.2-pro","url":"https://openrouter.ai/black-forest-labs/flux.2-pro","free":true,"batch":false}],"benchmarks":[{"benchmark":"AA Image Arena Elo","score":1207,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/image/leaderboard/text-to-image","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[]},{"id":"black-forest-labs/flux.2-flex","name":"FLUX.2 Flex","releaseLabel":".2 Flex","generationId":"releases","generationLabel":"Releases","variantLabel":".2 Flex","createdAt":"2025-11-25T04:46:27.000Z","kind":"image","contextLength":67344,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":14.6484375,"supportsTools":false,"supportsReasoning":false,"description":"FLUX.2 [flex] excels at rendering complex text, typography, and fine details, and supports multi-reference editing in the same unified architecture. Pricing is as follows, [per the docs](https://bfl.ai/pricing?category=flux.2): We charge $0.06...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"black-forest-labs/flux.2-flex","url":"https://openrouter.ai/black-forest-labs/flux.2-flex","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[]},{"id":"black-forest-labs/flux.2-max","name":"FLUX.2 Max","releaseLabel":".2 Max","generationId":"releases","generationLabel":"Releases","variantLabel":".2 Max","createdAt":"2025-12-16T03:59:30.000Z","kind":"image","contextLength":46864,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":17.08984375,"supportsTools":false,"supportsReasoning":false,"description":"FLUX.2 [max] is the new top-tier image model from Black Forest Labs, pushing image quality, prompt understanding, and editing consistency to the highest level yet. Pricing is as follows, [per...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"black-forest-labs/flux.2-max","url":"https://openrouter.ai/black-forest-labs/flux.2-max","free":true,"batch":false}],"benchmarks":[{"benchmark":"AA Image Arena Elo","score":1227,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/image/leaderboard/text-to-image","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[]},{"id":"black-forest-labs/flux.2-klein-4b","name":"FLUX.2 Klein 4B","releaseLabel":".2 Klein 4B","generationId":"4b","generationLabel":"4B","variantLabel":".2 Klein","createdAt":"2026-01-14T22:20:28.000Z","kind":"image","contextLength":40960,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":3.41796875,"supportsTools":false,"supportsReasoning":false,"description":"FLUX.2 [klein] 4B is the fastest and most cost-effective model in the FLUX.2 family, optimized for high-throughput use cases while maintaining excellent image quality. Pricing is based on the output...","huggingFaceId":"black-forest-labs/FLUX.2-klein-4B","routes":[{"provider":"OpenRouter","modelId":"black-forest-labs/flux.2-klein-4b","url":"https://openrouter.ai/black-forest-labs/flux.2-klein-4b","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[]}]},{"id":"flux-video","label":"FLUX Video","models":[{"id":"black-forest-labs/flux-3-video-20260804","name":"FLUX.3 Video","releaseLabel":"FLUX.3 Video","generationId":"3","generationLabel":"3","variantLabel":"Flux 3 Video","createdAt":"2026-08-04T15:53:51.000Z","kind":"video","contextLength":0,"inputModalities":["text","image","video"],"outputModalities":["video"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"FLUX.3 Video is a video generation model from Black Forest Labs. It supports text-to-video, image-guided generation with opening and closing keyframes, and video continuation workflows, making it suited for controlled...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"black-forest-labs/flux-3-video","url":"https://openrouter.ai/black-forest-labs/flux-3-video","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Video Arena","url":"https://artificialanalysis.ai/video/leaderboard/text-to-video","note":"Blind preference ranking for video generation"}],"intel":[]},{"id":"black-forest-labs/flux-video-upscale-20260819","name":"FLUX Video Upscale","releaseLabel":"Upscale","generationId":"releases","generationLabel":"Releases","variantLabel":"Upscale","createdAt":"2026-08-19T21:11:59.000Z","kind":"video","contextLength":0,"inputModalities":["text","video"],"outputModalities":["video"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"FLUX Video Upscale is a video upscaling model from Black Forest Labs. It enlarges a single source video by 1.5× to 3× while preserving its duration, with an optional prompt...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"black-forest-labs/flux-video-upscale","url":"https://openrouter.ai/black-forest-labs/flux-video-upscale","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Video Arena","url":"https://artificialanalysis.ai/video/leaderboard/text-to-video","note":"Blind preference ranking for video generation"}],"intel":[]}]}]},{"id":"bytedance","name":"ByteDance","domain":"bytedance.com","major":false,"modelCount":6,"series":[{"id":"seedance","label":"Seedance","models":[{"id":"bytedance/seedance-1-5-pro-20260320","name":"Seedance 1.5 Pro","releaseLabel":"1.5 Pro","generationId":"1","generationLabel":"1","variantLabel":"5 Pro","createdAt":"2026-03-23T14:53:28.000Z","kind":"video","contextLength":0,"inputModalities":["text","image"],"outputModalities":["video"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"ByteDance's next-generation audio-visual generation model with a 4.5B parameter Dual-Branch Diffusion Transformer architecture. Seedance 1.5 Pro generates video and audio simultaneously in a single unified pass — eliminating the timing...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"bytedance/seedance-1-5-pro","url":"https://openrouter.ai/bytedance/seedance-1-5-pro","free":true,"batch":false}],"benchmarks":[{"benchmark":"AA Video Arena Elo","score":1000,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/video/leaderboard/text-to-video","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Video Arena","url":"https://artificialanalysis.ai/video/leaderboard/text-to-video","note":"Blind preference ranking for video generation"}],"intel":[]},{"id":"bytedance/seedance-2.0-20260414","name":"Seedance 2.0","releaseLabel":"2.0","generationId":"2.0","generationLabel":"2.0","variantLabel":"Base","createdAt":"2026-04-15T00:02:42.000Z","kind":"video","contextLength":0,"inputModalities":["text","image","video","audio"],"outputModalities":["video"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Seedance 2.0 is a video generation model from ByteDance. It supports text-to-video, image-to-video with first and last frame control, and multimodal reference-to-video. It is particularly strong at preserving character consistency,...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"bytedance/seedance-2.0","url":"https://openrouter.ai/bytedance/seedance-2.0","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Video Arena","url":"https://artificialanalysis.ai/video/leaderboard/text-to-video","note":"Blind preference ranking for video generation"}],"intel":[]},{"id":"bytedance/seedance-2.0-fast-20260414","name":"Seedance 2.0 Fast","releaseLabel":"2.0 Fast","generationId":"2.0","generationLabel":"2.0","variantLabel":"Fast","createdAt":"2026-04-15T00:02:42.000Z","kind":"video","contextLength":0,"inputModalities":["text","image","video","audio"],"outputModalities":["video"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Seedance 2.0 Fast is a video generation model from ByteDance. It supports text-to-video, image-to-video with first and last frame control, and multimodal reference-to-video. It prioritizes generation speed and lower cost...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"bytedance/seedance-2.0-fast","url":"https://openrouter.ai/bytedance/seedance-2.0-fast","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Video Arena","url":"https://artificialanalysis.ai/video/leaderboard/text-to-video","note":"Blind preference ranking for video generation"}],"intel":[]},{"id":"bytedance/seedance-2.5-20260807","name":"Seedance 2.5","releaseLabel":"2.5","generationId":"2.5","generationLabel":"2.5","variantLabel":"Base","createdAt":"2026-08-07T22:20:53.000Z","kind":"video","contextLength":0,"inputModalities":["text","image","video","audio"],"outputModalities":["video"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Seedance 2.5 is a video generation model from ByteDance. It is suited for long-form storytelling, multimodal reference-based generation, video editing, and video extension. It supports first-frame and first-and-last-frame control, up...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"bytedance/seedance-2.5","url":"https://openrouter.ai/bytedance/seedance-2.5","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Video Arena","url":"https://artificialanalysis.ai/video/leaderboard/text-to-video","note":"Blind preference ranking for video generation"}],"intel":[]},{"id":"bytedance/seedance-2.0-mini-20260811","name":"Seedance 2.0 Mini","releaseLabel":"2.0 Mini","generationId":"2.0","generationLabel":"2.0","variantLabel":"Mini","createdAt":"2026-08-12T16:36:40.000Z","kind":"video","contextLength":0,"inputModalities":["text","image","video","audio"],"outputModalities":["video"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Seedance 2.0 Mini is a video generation model from ByteDance. It supports text-to-video, image-to-video with first and last frame control, and multimodal reference-to-video with image, video, and audio inputs. It...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"bytedance/seedance-2.0-mini","url":"https://openrouter.ai/bytedance/seedance-2.0-mini","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Video Arena","url":"https://artificialanalysis.ai/video/leaderboard/text-to-video","note":"Blind preference ranking for video generation"}],"intel":[]}]},{"id":"ui","label":"UI","models":[{"id":"bytedance/ui-tars-1.5-7b","name":"UI-TARS 7B","releaseLabel":"TARS 7B","generationId":"1.5","generationLabel":"1.5","variantLabel":"7B","createdAt":"2025-07-22T17:24:16.000Z","kind":"multimodal","contextLength":128000,"inputModalities":["image","text"],"outputModalities":["text"],"promptPerMillion":0.09999999999999999,"completionPerMillion":0.19999999999999998,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"UI-TARS-1.5 is a multimodal vision-language agent optimized for GUI-based environments, including desktop interfaces, web browsers, mobile systems, and games. Built by ByteDance, it builds upon the UI-TARS framework with reinforcement...","huggingFaceId":"ByteDance-Seed/UI-TARS-1.5-7B","routes":[{"provider":"OpenRouter","modelId":"bytedance/ui-tars-1.5-7b","url":"https://openrouter.ai/bytedance/ui-tars-1.5-7b","free":false,"batch":false}],"benchmarks":[{"benchmark":"ScreenSpot-Pro · overall","score":61.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://gui-agent.github.io/grounding-leaderboard/","measuredAt":"2026-03-17","scope":"Registered model evaluation · ScreenSpot-Pro Leaderboard"},{"benchmark":"ScreenSpot-Pro · android studio macos","score":57.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://gui-agent.github.io/grounding-leaderboard/","measuredAt":"2026-03-17","scope":"Registered model evaluation · ScreenSpot-Pro Leaderboard"},{"benchmark":"ScreenSpot-Pro · autocad windows","score":41.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://gui-agent.github.io/grounding-leaderboard/","measuredAt":"2026-03-17","scope":"Registered model evaluation · ScreenSpot-Pro Leaderboard"},{"benchmark":"ScreenSpot-Pro · blender windows","score":59.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://gui-agent.github.io/grounding-leaderboard/","measuredAt":"2026-03-17","scope":"Registered model evaluation · ScreenSpot-Pro Leaderboard"},{"benchmark":"ScreenSpot-Pro · davinci macos","score":54.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://gui-agent.github.io/grounding-leaderboard/","measuredAt":"2026-03-17","scope":"Registered model evaluation · ScreenSpot-Pro Leaderboard"},{"benchmark":"ScreenSpot-Pro · eviews windows","score":96,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://gui-agent.github.io/grounding-leaderboard/","measuredAt":"2026-03-17","scope":"Registered model evaluation · ScreenSpot-Pro Leaderboard"},{"benchmark":"ScreenSpot-Pro · excel macos","score":68.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://gui-agent.github.io/grounding-leaderboard/","measuredAt":"2026-03-17","scope":"Registered model evaluation · ScreenSpot-Pro Leaderboard"},{"benchmark":"ScreenSpot-Pro · fruitloops windows","score":43.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://gui-agent.github.io/grounding-leaderboard/","measuredAt":"2026-03-17","scope":"Registered model evaluation · ScreenSpot-Pro Leaderboard"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"f7f80504-c539-4937-a09e-7ff21d0a67fc","title":"ChatDev 2.0","href":"/intel/f7f80504-c539-4937-a09e-7ff21d0a67fc","hook":"Zero-code multi-agent orchestration: you describe the system in config and it builds and runs the agent team. Worth a look if you are tired of hand-wiring ag…"},{"id":"e4da4983-d8f3-41b2-b60b-48fab0eadc75","title":"Agent Memory Architecture Guide","href":"/intel/e4da4983-d8f3-41b2-b60b-48fab0eadc75","hook":"A technical progression from a Python list to graph-vector hybrid memory, with the tradeoffs at each step. The reference to reach for when \"just stuff it in …"},{"id":"db21ae54-dd87-4cf7-8fb4-62c8fad8730e","title":"Claude Token Optimization Guide","href":"/intel/db21ae54-dd87-4cf7-8fb4-62c8fad8730e","hook":"Nine concrete tactics to cut Claude token usage and stop hitting limits — prompt editing, batching, context discipline. Practical and immediately applicable …"}]}]}]},{"id":"openrouter","name":"OpenRouter","domain":"openrouter.ai","major":false,"modelCount":6,"series":[{"id":"auto","label":"Auto","models":[{"id":"openrouter/auto","name":"Auto Router","releaseLabel":"Router","generationId":"releases","generationLabel":"Releases","variantLabel":"Router","createdAt":"2023-11-08T00:00:00.000Z","kind":"image","contextLength":2000000,"inputModalities":["text","image","audio","file","video"],"outputModalities":["text","image"],"promptPerMillion":-1000000,"completionPerMillion":-1000000,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Your prompt will be processed by a meta-model and routed to one of dozens of models (see below), optimizing for the best possible output. To see which model was used,...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"openrouter/auto","url":"https://openrouter.ai/openrouter/auto","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[{"id":"a8358232-73db-4250-b641-8826009c7e68","title":"AI Autoresearch Strategy Generator","href":"/intel/a8358232-73db-4250-b641-8826009c7e68","hook":"Parallel Claude agents that generate, test, and optimize trading strategies — and won a hackathon doing it. Read it for the parallel-agent search pattern, wh…"},{"id":"9d1e4b51-a860-48ed-ad4d-a02ae6167be0","title":"What Happened When I Applied Karpathy's Autoresearch Idea to LLM Inference","href":"/intel/9d1e4b51-a860-48ed-ad4d-a02ae6167be0","hook":"Most optimization demos show you the win and hide the search; this one shows the search. An honest walkthrough of applying Karpathy's autoresearch idea to LL…"},{"id":"dd3c95ad-0fac-49bd-9582-e0c78686e14f","title":"AutoAgent: Self-Optimizing AI Agents","href":"/intel/dd3c95ad-0fac-49bd-9582-e0c78686e14f","hook":"An open-source meta-agent that tunes task agents — prompts, tools, orchestration — until performance climbs, no human in the loop. A working reference if you…"}]},{"id":"openrouter/auto-beta","name":"Auto Router (Beta)","releaseLabel":"Router (Beta)","generationId":"releases","generationLabel":"Releases","variantLabel":"Router (beta)","createdAt":"2026-07-17T17:59:25.000Z","kind":"image","contextLength":2000000,"inputModalities":["text","image","audio","file","video"],"outputModalities":["text","image"],"promptPerMillion":-1000000,"completionPerMillion":-1000000,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Auto Router (Beta) is a task-aware router from OpenRouter. It classifies each request, then routes it the [most popular model](/rankings#task-spend) for that task based on aggregate spend, filtered by your...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"openrouter/auto-beta","url":"https://openrouter.ai/openrouter/auto-beta","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[{"id":"a8358232-73db-4250-b641-8826009c7e68","title":"AI Autoresearch Strategy Generator","href":"/intel/a8358232-73db-4250-b641-8826009c7e68","hook":"Parallel Claude agents that generate, test, and optimize trading strategies — and won a hackathon doing it. Read it for the parallel-agent search pattern, wh…"},{"id":"9d1e4b51-a860-48ed-ad4d-a02ae6167be0","title":"What Happened When I Applied Karpathy's Autoresearch Idea to LLM Inference","href":"/intel/9d1e4b51-a860-48ed-ad4d-a02ae6167be0","hook":"Most optimization demos show you the win and hide the search; this one shows the search. An honest walkthrough of applying Karpathy's autoresearch idea to LL…"},{"id":"dd3c95ad-0fac-49bd-9582-e0c78686e14f","title":"AutoAgent: Self-Optimizing AI Agents","href":"/intel/dd3c95ad-0fac-49bd-9582-e0c78686e14f","hook":"An open-source meta-agent that tunes task agents — prompts, tools, orchestration — until performance climbs, no human in the loop. A working reference if you…"}]}]},{"id":"bodybuilder","label":"Bodybuilder","models":[{"id":"openrouter/bodybuilder","name":"Body Builder (beta)","releaseLabel":"Body Builder (beta)","generationId":"releases","generationLabel":"Releases","variantLabel":"Body Builder (beta)","createdAt":"2025-12-05T03:00:53.000Z","kind":"language","contextLength":128000,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":-1000000,"completionPerMillion":-1000000,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Transform your natural language requests into structured OpenRouter API request objects. Describe what you want to accomplish with AI models, and Body Builder will construct the appropriate API calls. Example:...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openrouter/bodybuilder","url":"https://openrouter.ai/openrouter/bodybuilder","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]},{"id":"free","label":"Free","models":[{"id":"openrouter/free","name":"Free Models Router","releaseLabel":"Models Router","generationId":"releases","generationLabel":"Releases","variantLabel":"Models Router","createdAt":"2026-02-01T03:43:47.000Z","kind":"multimodal","contextLength":200000,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"The simplest way to get free inference. openrouter/free is a router that selects free models at random from the models available on OpenRouter. The router smartly filters for models that...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openrouter/free","url":"https://openrouter.ai/openrouter/free","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"cefdb845-e661-43e5-8120-6ae2ed0fe4e5","title":"Learning Word Embedding","href":"/intel/cefdb845-e661-43e5-8120-6ae2ed0fe4e5","hook":"A clear walkthrough of how free-text words become numeric vectors — from one-hot encoding to learned embeddings — grounding the intuition behind the embeddin…"},{"id":"081a5993-5053-420d-b327-9d8adf796046","title":"RLHF Book & Post-training Course","href":"/intel/081a5993-5053-420d-b327-9d8adf796046","hook":"If you're moving from prompting into actually shaping model behavior, this pairs a professionally edited RLHF book with a free structured lecture course cove…"},{"id":"498a09a7-c6c9-46b8-8b99-7afb709d9336","title":"An Open Course on LLMs, Led by Practitioners","href":"/intel/498a09a7-c6c9-46b8-8b99-7afb709d9336","hook":"A free, well-organized 40+ hour course distilled from a popular paid program, with annotated talks and notes from practitioners across evals, RAG, and fine-t…"}]}]},{"id":"fusion","label":"Fusion","models":[{"id":"openrouter/fusion","name":"Fusion","releaseLabel":"Fusion","generationId":"releases","generationLabel":"Releases","variantLabel":"Fusion","createdAt":"2026-06-13T17:27:27.000Z","kind":"language","contextLength":1000000,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":-1000000,"completionPerMillion":-1000000,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Fusion turns your prompt into a small multi-model deliberation. A panel of expert models (see below) analyzes your prompt in parallel with web search and web fetch enabled, then a...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"openrouter/fusion","url":"https://openrouter.ai/openrouter/fusion","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"86ad41e0-d68f-4de9-bb40-7aeebc3ffd1b","title":"Diffusion Models for Video Generation","href":"/intel/86ad41e0-d68f-4de9-bb40-7aeebc3ffd1b","hook":"A structured survey of how diffusion models extend from image to video generation, unpacking the temporal-consistency and data-scarcity problems that define …"},{"id":"3dc89a32-e53d-49b7-a33a-1acbe1822b33","title":"What are Diffusion Models?","href":"/intel/3dc89a32-e53d-49b7-a33a-1acbe1822b33","hook":"A rigorous, continuously-updated technical walkthrough of diffusion models — from the DDPM math to consistency models and latent diffusion — that gives engin…"}]}]},{"id":"pareto","label":"Pareto","models":[{"id":"openrouter/pareto-code","name":"Pareto Code Router","releaseLabel":"Code Router","generationId":"releases","generationLabel":"Releases","variantLabel":"Code Router","createdAt":"2026-04-21T05:05:00.000Z","kind":"language","contextLength":2000000,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":-1000000,"completionPerMillion":-1000000,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"The Pareto Router maintains a tiered shortlist of strong coding models, ranked by [Artificial Analysis](https://artificialanalysis.ai/) coding percentiles. Set min_coding_score between 0 and 1 on the [pareto-router plugin](https://openrouter.ai/docs/guides/routing/routers/pareto-router#the-min_coding_score-parameter) to control how...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"openrouter/pareto-code","url":"https://openrouter.ai/openrouter/pareto-code","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]}]},{"id":"tencent","name":"Tencent","domain":"tencent.com","major":false,"modelCount":6,"series":[{"id":"hy","label":"HY","models":[{"id":"tencent/hy-mt2-7b-20260521","name":"Hy-MT2-7B","releaseLabel":"MT2-7B","generationId":"7b","generationLabel":"7B","variantLabel":"Mt2","createdAt":"2026-08-19T14:13:17.000Z","kind":"language","contextLength":8192,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.074,"completionPerMillion":0.295,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Hy-MT2-7B is a 7B-parameter translation model from Tencent. It supports 33 language pairs and five Chinese dialect and minority-language pairs, with workflows for structured, delimiter-based, contextual, glossary-based, and style-guided translation.","huggingFaceId":"tencent/Hy-MT2-7B","routes":[{"provider":"OpenRouter","modelId":"tencent/hy-mt2-7b","url":"https://openrouter.ai/tencent/hy-mt2-7b","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"e4da4983-d8f3-41b2-b60b-48fab0eadc75","title":"Agent Memory Architecture Guide","href":"/intel/e4da4983-d8f3-41b2-b60b-48fab0eadc75","hook":"A technical progression from a Python list to graph-vector hybrid memory, with the tradeoffs at each step. The reference to reach for when \"just stuff it in …"},{"id":"d368d9a4-37d9-456a-bc0c-081ae8dd4c28","title":"The SaaS Reckoning","href":"/intel/d368d9a4-37d9-456a-bc0c-081ae8dd4c28","hook":"How stock-based comp has masked the true costs of SaaS, and why AI disruption forces the bill due. A financially literate take for founders and operators who…"},{"id":"49e0adc5-ae09-46c8-8d10-582885877804","title":"Agent Harness Memory Control · article","href":"/intel/49e0adc5-ae09-46c8-8d10-582885877804","hook":"Why the harness — and specifically who controls your agent's memory — is critical infrastructure, and why open harnesses matter. A short, sharp argument for …"}]},{"id":"tencent/hy-mt2-30b-a3b-20260521","name":"Hy-MT2-30B-A3B","releaseLabel":"MT2-30B-A3B","generationId":"30b","generationLabel":"30B","variantLabel":"A3B","createdAt":"2026-08-20T13:12:41.000Z","kind":"language","contextLength":8192,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.074,"completionPerMillion":0.295,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Hy-MT2-30B-A3B is Tencent's flagship translation model in the Hy-MT2 family. It supports 33 language pairs and five Chinese dialect and minority-language pairs, with workflows for structured, delimiter-based, contextual, glossary-based, and...","huggingFaceId":"tencent/Hy-MT2-30B-A3B","routes":[{"provider":"OpenRouter","modelId":"tencent/hy-mt2-30b-a3b","url":"https://openrouter.ai/tencent/hy-mt2-30b-a3b","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"e4da4983-d8f3-41b2-b60b-48fab0eadc75","title":"Agent Memory Architecture Guide","href":"/intel/e4da4983-d8f3-41b2-b60b-48fab0eadc75","hook":"A technical progression from a Python list to graph-vector hybrid memory, with the tradeoffs at each step. The reference to reach for when \"just stuff it in …"},{"id":"d368d9a4-37d9-456a-bc0c-081ae8dd4c28","title":"The SaaS Reckoning","href":"/intel/d368d9a4-37d9-456a-bc0c-081ae8dd4c28","hook":"How stock-based comp has masked the true costs of SaaS, and why AI disruption forces the bill due. A financially literate take for founders and operators who…"},{"id":"49e0adc5-ae09-46c8-8d10-582885877804","title":"Agent Harness Memory Control · article","href":"/intel/49e0adc5-ae09-46c8-8d10-582885877804","hook":"Why the harness — and specifically who controls your agent's memory — is critical infrastructure, and why open harnesses matter. A short, sharp argument for …"}]},{"id":"tencent/hy-mt2-1.8b-20260521","name":"Hy-MT2-1.8B","releaseLabel":"MT2-1.8B","generationId":"1.8b","generationLabel":"1.8B","variantLabel":"Mt2","createdAt":"2026-08-20T13:13:01.000Z","kind":"language","contextLength":8192,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.044,"completionPerMillion":0.17700000000000002,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Hy-MT2-1.8B is a compact 1.8B-parameter translation model from Tencent. It supports 33 language pairs and five Chinese dialect and minority-language pairs, with workflows for structured, delimiter-based, contextual, glossary-based, and style-guided...","huggingFaceId":"tencent/Hy-MT2-1.8B","routes":[{"provider":"OpenRouter","modelId":"tencent/hy-mt2-1.8b","url":"https://openrouter.ai/tencent/hy-mt2-1.8b","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"e4da4983-d8f3-41b2-b60b-48fab0eadc75","title":"Agent Memory Architecture Guide","href":"/intel/e4da4983-d8f3-41b2-b60b-48fab0eadc75","hook":"A technical progression from a Python list to graph-vector hybrid memory, with the tradeoffs at each step. The reference to reach for when \"just stuff it in …"},{"id":"d368d9a4-37d9-456a-bc0c-081ae8dd4c28","title":"The SaaS Reckoning","href":"/intel/d368d9a4-37d9-456a-bc0c-081ae8dd4c28","hook":"How stock-based comp has masked the true costs of SaaS, and why AI disruption forces the bill due. A financially literate take for founders and operators who…"},{"id":"49e0adc5-ae09-46c8-8d10-582885877804","title":"Agent Harness Memory Control · article","href":"/intel/49e0adc5-ae09-46c8-8d10-582885877804","hook":"Why the harness — and specifically who controls your agent's memory — is critical infrastructure, and why open harnesses matter. A short, sharp argument for …"}]}]},{"id":"hy3","label":"HY3","models":[{"id":"tencent/hy3-preview-20260421","name":"Hy3 preview","releaseLabel":"preview","generationId":"releases","generationLabel":"Releases","variantLabel":"Preview","createdAt":"2026-04-22T17:15:50.000Z","kind":"language","contextLength":262144,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.18,"completionPerMillion":0.6,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Hy3 preview is a high-efficiency Mixture-of-Experts model from Tencent designed for agentic workflows and production use. It supports configurable reasoning levels across disabled, low, and high modes, allowing it to...","huggingFaceId":"tencent/Hy3-preview","routes":[{"provider":"OpenRouter","modelId":"tencent/hy3-preview","url":"https://openrouter.ai/tencent/hy3-preview","free":false,"batch":false}],"benchmarks":[{"benchmark":"AA Intelligence Index","score":42.2,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/hy3","measuredAt":"2026-08-27","scope":"Current public index; highest tested effort configuration"},{"benchmark":"AA Agentic Index","score":31.4,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/hy3","measuredAt":"2026-08-27","scope":"Current public agentic composite"},{"benchmark":"SciCode","score":47.6,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/hy3","measuredAt":"2026-08-27","scope":"Scientific coding evaluation"},{"benchmark":"AA Omniscience Index","score":-18.5,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/hy3","measuredAt":"2026-08-27","scope":"Current public knowledge and hallucination index"},{"benchmark":"GDPval-AA normalized","score":35.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/hy3","measuredAt":"2026-08-27","scope":"Normalized real-world knowledge-work evaluation"},{"benchmark":"τ²-bench Banking","score":22.9,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/hy3","measuredAt":"2026-08-27","scope":"Tool-agent benchmark in the banking domain"},{"benchmark":"Terminal-Bench 2.1","score":64.4,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/hy3","measuredAt":"2026-08-27","scope":"Terminal-agent task completion"},{"benchmark":"LCR","score":74.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/hy3","measuredAt":"2026-08-27","scope":"Long-context reasoning evaluation"},{"benchmark":"Humanity's Last Exam","score":33.5,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/hy3","measuredAt":"2026-08-27","scope":"Broad expert-level knowledge evaluation"},{"benchmark":"GPQA","score":89.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/hy3","measuredAt":"2026-08-27","scope":"Graduate-level science questions"},{"benchmark":"CritPt","score":4.9,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/hy3","measuredAt":"2026-08-27","scope":"Critical-point reasoning evaluation"},{"benchmark":"LMArena overall","score":1441,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 60 · 5,877 votes"},{"benchmark":"gpqa · diamond","score":87.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/tencent/Hy3-preview","measuredAt":"2026-04-14","scope":"Registered model evaluation · Model Card"},{"benchmark":"hle · hle","score":30,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/tencent/Hy3-preview","measuredAt":"2026-04-14","scope":"Registered model evaluation · Model Card"},{"benchmark":"SWE-bench_Verified · swe bench % resolved","score":74.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/tencent/Hy3-preview","measuredAt":"2026-04-14","scope":"Registered model evaluation · Model Card"},{"benchmark":"terminal-bench-2.0 · terminalbench 2","score":54.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/tencent/Hy3-preview","measuredAt":"2026-04-14","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"tencent/hy3-20260706","name":"Hy3","releaseLabel":"Hy3","generationId":"releases","generationLabel":"Releases","variantLabel":"Hy3","createdAt":"2026-07-06T13:20:48.000Z","kind":"language","contextLength":262144,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.0825,"completionPerMillion":0.33,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...","huggingFaceId":"tencent/Hy3","routes":[{"provider":"OpenRouter","modelId":"tencent/hy3","url":"https://openrouter.ai/tencent/hy3","free":false,"batch":false}],"benchmarks":[{"benchmark":"AA Intelligence Index","score":42.2,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/hy3","measuredAt":"2026-08-27","scope":"Current public index; highest tested effort configuration"},{"benchmark":"AA Agentic Index","score":31.4,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/hy3","measuredAt":"2026-08-27","scope":"Current public agentic composite"},{"benchmark":"SciCode","score":47.6,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/hy3","measuredAt":"2026-08-27","scope":"Scientific coding evaluation"},{"benchmark":"AA Omniscience Index","score":-18.5,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/hy3","measuredAt":"2026-08-27","scope":"Current public knowledge and hallucination index"},{"benchmark":"GDPval-AA normalized","score":35.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/hy3","measuredAt":"2026-08-27","scope":"Normalized real-world knowledge-work evaluation"},{"benchmark":"τ²-bench Banking","score":22.9,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/hy3","measuredAt":"2026-08-27","scope":"Tool-agent benchmark in the banking domain"},{"benchmark":"Terminal-Bench 2.1","score":64.4,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/hy3","measuredAt":"2026-08-27","scope":"Terminal-agent task completion"},{"benchmark":"LCR","score":74.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/hy3","measuredAt":"2026-08-27","scope":"Long-context reasoning evaluation"},{"benchmark":"Humanity's Last Exam","score":33.5,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/hy3","measuredAt":"2026-08-27","scope":"Broad expert-level knowledge evaluation"},{"benchmark":"GPQA","score":89.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/hy3","measuredAt":"2026-08-27","scope":"Graduate-level science questions"},{"benchmark":"CritPt","score":4.9,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/hy3","measuredAt":"2026-08-27","scope":"Critical-point reasoning evaluation"},{"benchmark":"LMArena overall","score":1441,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 60 · 5,877 votes"},{"benchmark":"terminal-bench-2.1 · terminalbench 2 1","score":71.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/tencent/Hy3","measuredAt":"2026-07-20","scope":"Registered model evaluation · Model Card"},{"benchmark":"SWE-bench_Multilingual · swe bench multilingual % resolved","score":75.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/tencent/Hy3","measuredAt":"2026-08-10","scope":"Registered model evaluation · Model Card"},{"benchmark":"Long-Horizon-Terminal-Bench · lhtb solved","score":1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://zli12321.github.io/LHTB/leaderboard.html","measuredAt":"2026-07-16","scope":"Registered model evaluation · LHTB leaderboard"},{"benchmark":"apex-agents · apex-agents","score":25.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/tencent/Hy3","measuredAt":"2026-07-06","scope":"Registered model evaluation · Model Card"},{"benchmark":"deep-swe · deep swe","score":28,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/tencent/Hy3","measuredAt":"2026-07-06","scope":"Registered model evaluation · Model Card"},{"benchmark":"gpqa · diamond","score":90.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/tencent/Hy3","measuredAt":"2026-07-06","scope":"Registered model evaluation · Model Card"},{"benchmark":"hle · hle","score":53.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/tencent/Hy3","measuredAt":"2026-07-06","scope":"Registered model evaluation · Model Card"},{"benchmark":"skillsbench · skillsbench v1 1","score":55.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/tencent/Hy3","measuredAt":"2026-07-06","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]},{"id":"hunyuan","label":"Hunyuan","models":[{"id":"tencent/hunyuan-a13b-instruct","name":"Hunyuan A13B Instruct","releaseLabel":"A13B Instruct","generationId":"releases","generationLabel":"Releases","variantLabel":"A13B Instruct","createdAt":"2025-07-08T15:14:24.000Z","kind":"language","contextLength":131072,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.14,"completionPerMillion":0.5700000000000001,"imageOutput":null,"supportsTools":false,"supportsReasoning":true,"description":"Hunyuan-A13B is a 13B active parameter Mixture-of-Experts (MoE) language model developed by Tencent, with a total parameter count of 80B and support for reasoning via Chain-of-Thought. It offers competitive benchmark...","huggingFaceId":"tencent/Hunyuan-A13B-Instruct","routes":[{"provider":"OpenRouter","modelId":"tencent/hunyuan-a13b-instruct","url":"https://openrouter.ai/tencent/hunyuan-a13b-instruct","free":false,"batch":false}],"benchmarks":[{"benchmark":"MMLU-Pro · mmlu pro","score":67.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/datasets/evaleval/EEE_datastore/blob/192329fb7d6b15b7b0936a1a58ae862aa7e8ba24/flat/objects/4b/ff/4bffdd42-60c3-41a9-9ef8-e5849e114d9e.json","measuredAt":"2026-06-30","scope":"Registered model evaluation · EvalEval"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]}]},{"id":"amazon","name":"Amazon","domain":"amazon.com","major":false,"modelCount":5,"series":[{"id":"nova","label":"Nova","models":[{"id":"amazon/nova-pro-v1","name":"Nova Pro 1.0","releaseLabel":"Pro 1.0","generationId":"v1","generationLabel":"V1","variantLabel":"Pro 1.0","createdAt":"2024-12-05T22:05:03.000Z","kind":"multimodal","contextLength":300000,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0.7999999999999999,"completionPerMillion":3.1999999999999997,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Amazon Nova Pro 1.0 is a capable multimodal model from Amazon focused on providing a combination of accuracy, speed, and cost for a wide range of tasks. As of December...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"amazon/nova-pro-v1","url":"https://openrouter.ai/amazon/nova-pro-v1","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"}]},{"id":"amazon/nova-micro-v1","name":"Nova Micro 1.0","releaseLabel":"Micro 1.0","generationId":"v1","generationLabel":"V1","variantLabel":"Micro 1.0","createdAt":"2024-12-05T22:20:37.000Z","kind":"language","contextLength":128000,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.035,"completionPerMillion":0.14,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Amazon Nova Micro 1.0 is a text-only model that delivers the lowest latency responses in the Amazon Nova family of models at a very low cost. With a context length...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"amazon/nova-micro-v1","url":"https://openrouter.ai/amazon/nova-micro-v1","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"}]},{"id":"amazon/nova-lite-v1","name":"Nova Lite 1.0","releaseLabel":"Lite 1.0","generationId":"v1","generationLabel":"V1","variantLabel":"Lite 1.0","createdAt":"2024-12-05T22:22:43.000Z","kind":"multimodal","contextLength":300000,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0.06,"completionPerMillion":0.24,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Amazon Nova Lite 1.0 is a very low-cost multimodal model from Amazon that focused on fast processing of image, video, and text inputs to generate text output. Amazon Nova Lite...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"amazon/nova-lite-v1","url":"https://openrouter.ai/amazon/nova-lite-v1","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"}]},{"id":"amazon/nova-premier-v1","name":"Nova Premier 1.0","releaseLabel":"Premier 1.0","generationId":"v1","generationLabel":"V1","variantLabel":"Premier 1.0","createdAt":"2025-10-31T22:38:52.000Z","kind":"multimodal","contextLength":1000000,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":2.5,"completionPerMillion":12.5,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Amazon Nova Premier is the most capable of Amazon’s multimodal models for complex reasoning tasks and for use as the best teacher for distilling custom models.","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"amazon/nova-premier-v1","url":"https://openrouter.ai/amazon/nova-premier-v1","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"}]},{"id":"amazon/nova-2-lite-v1","name":"Nova 2 Lite","releaseLabel":"2 Lite","generationId":"2","generationLabel":"2","variantLabel":"Lite V1","createdAt":"2025-12-02T17:31:12.000Z","kind":"multimodal","contextLength":1000000,"inputModalities":["text","image","video","file"],"outputModalities":["text"],"promptPerMillion":0.3,"completionPerMillion":2.5,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Nova 2 Lite is a fast, cost-effective reasoning model for everyday workloads that can process text, images, and videos to generate text. Nova 2 Lite demonstrates standout capabilities in processing...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"amazon/nova-2-lite-v1","url":"https://openrouter.ai/amazon/nova-2-lite-v1","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1362,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 177 · 12,294 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"0ff56575-5aab-4462-94f0-38573203b69b","title":"Why most AI products fail: Lessons from 50+ AI deployments at OpenAI, Google, and Amazon","href":"/intel/0ff56575-5aab-4462-94f0-38573203b69b","hook":"If you're building AI-native products, this distills real deployment lessons from OpenAI/Google/Amazon veterans into a concrete iterative framework — includi…"}]}]}]},{"id":"fish-audio","name":"Fish Audio","domain":"fish.audio","major":false,"modelCount":5,"series":[{"id":"s2.1","label":"S2.1","models":[{"id":"fish-audio/s2.1-pro-20260729","name":"S2.1 Pro","releaseLabel":"Pro","generationId":"releases","generationLabel":"Releases","variantLabel":"Pro","createdAt":"2026-07-29T19:35:32.000Z","kind":"audio","contextLength":0,"inputModalities":["text"],"outputModalities":["speech"],"promptPerMillion":15,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"S2.1 Pro is a production-oriented text-to-speech model from Fish Audio. It is suited for multilingual voice applications, expressive narration, and dialogue synthesis, with open-ended natural-language controls for speaking style and...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"fish-audio/s2.1-pro","url":"https://openrouter.ai/fish-audio/s2.1-pro","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[]},{"id":"fish-audio/s2.1-pro-free-20260729","name":"S2.1 Pro Free","releaseLabel":"Pro Free","generationId":"releases","generationLabel":"Releases","variantLabel":"Pro Free","createdAt":"2026-07-29T19:35:33.000Z","kind":"audio","contextLength":0,"inputModalities":["text"],"outputModalities":["speech"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"S2.1 Pro Free is the no-cost variant of Fish Audio S2.1 Pro, intended for testing, prototyping, and low-volume applications. It provides the same synthesis capabilities without production latency or availability...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"fish-audio/s2.1-pro-free:free","url":"https://openrouter.ai/fish-audio/s2.1-pro-free:free","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[]}]},{"id":"s1","label":"S1","models":[{"id":"fish-audio/s1-20260729","name":"S1","releaseLabel":"S1","generationId":"releases","generationLabel":"Releases","variantLabel":"S1","createdAt":"2026-07-29T19:35:34.000Z","kind":"audio","contextLength":0,"inputModalities":["text"],"outputModalities":["speech"],"promptPerMillion":15,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"S1 is a multilingual text-to-speech model from Fish Audio. It is suited for voice applications that need broad emotional expression, using parenthetical controls to guide speaking style across its supported...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"fish-audio/s1","url":"https://openrouter.ai/fish-audio/s1","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[]}]},{"id":"s2","label":"S2","models":[{"id":"fish-audio/s2-pro-20260729","name":"S2 Pro","releaseLabel":"Pro","generationId":"releases","generationLabel":"Releases","variantLabel":"Pro","createdAt":"2026-07-29T19:35:34.000Z","kind":"audio","contextLength":0,"inputModalities":["text"],"outputModalities":["speech"],"promptPerMillion":15,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"S2 Pro is a multilingual text-to-speech model from Fish Audio. It is suited for expressive narration and multi-speaker dialogue, with natural-language controls for speaking style and emotion.","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"fish-audio/s2-pro","url":"https://openrouter.ai/fish-audio/s2-pro","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[]}]},{"id":"transcribe","label":"Transcribe","models":[{"id":"fish-audio/transcribe-1-20260729","name":"Transcribe 1","releaseLabel":"1","generationId":"1","generationLabel":"1","variantLabel":"Base","createdAt":"2026-07-29T19:35:35.000Z","kind":"audio","contextLength":0,"inputModalities":["audio"],"outputModalities":["transcription"],"promptPerMillion":100,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Transcribe 1 is a speech-to-text model from Fish Audio. It is suited for audio transcription with automatic language detection and can return timestamped word-level segments when alignment details are requested.","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"fish-audio/transcribe-1","url":"https://openrouter.ai/fish-audio/transcribe-1","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[]}]}]},{"id":"sentence-transformers","name":"Sentence Transformers","domain":"sbert.net","major":false,"modelCount":5,"series":[{"id":"all","label":"ALL","models":[{"id":"sentence-transformers/all-minilm-l6-v2-20251117","name":"all-MiniLM-L6-v2","releaseLabel":"MiniLM-L6-v2","generationId":"v2","generationLabel":"V2","variantLabel":"Minilm L6","createdAt":"2025-11-17T23:12:56.000Z","kind":"embedding","contextLength":512,"inputModalities":["text"],"outputModalities":["embeddings"],"promptPerMillion":0.005,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"The all-MiniLM-L6-v2 embedding model maps sentences and short paragraphs into a 384-dimensional dense vector space, enabling high-quality semantic representations that are ideal for downstream tasks such as information retrieval, clustering,...","huggingFaceId":"sentence-transformers/all-MiniLM-L6-v2","routes":[{"provider":"OpenRouter","modelId":"sentence-transformers/all-minilm-l6-v2","url":"https://openrouter.ai/sentence-transformers/all-minilm-l6-v2","free":false,"batch":false}],"benchmarks":[{"benchmark":"arguana · ArguAna default test","score":50.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-03-05","scope":"Registered model evaluation · Obtained using MTEB v1.12.75"},{"benchmark":"arguana · ArguAna","score":50.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-03-05","scope":"Registered model evaluation · Obtained using MTEB v1.12.75"},{"benchmark":"MTEB English retrieval","score":42.9,"unit":"%","sourceName":"MTEB","sourceUrl":"https://leaderboard.mteb.org/","measuredAt":"2026-08-27","scope":"Mean retrieval score · leaderboard rank 144"}],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[{"id":"a8358232-73db-4250-b641-8826009c7e68","title":"AI Autoresearch Strategy Generator","href":"/intel/a8358232-73db-4250-b641-8826009c7e68","hook":"Parallel Claude agents that generate, test, and optimize trading strategies — and won a hackathon doing it. Read it for the parallel-agent search pattern, wh…"},{"id":"d368d9a4-37d9-456a-bc0c-081ae8dd4c28","title":"The SaaS Reckoning","href":"/intel/d368d9a4-37d9-456a-bc0c-081ae8dd4c28","hook":"How stock-based comp has masked the true costs of SaaS, and why AI disruption forces the bill due. A financially literate take for founders and operators who…"},{"id":"49e0adc5-ae09-46c8-8d10-582885877804","title":"Agent Harness Memory Control · article","href":"/intel/49e0adc5-ae09-46c8-8d10-582885877804","hook":"Why the harness — and specifically who controls your agent's memory — is critical infrastructure, and why open harnesses matter. A short, sharp argument for …"}]},{"id":"sentence-transformers/all-mpnet-base-v2-20251117","name":"all-mpnet-base-v2","releaseLabel":"mpnet-base-v2","generationId":"v2","generationLabel":"V2","variantLabel":"Mpnet Base","createdAt":"2025-11-17T23:23:50.000Z","kind":"embedding","contextLength":512,"inputModalities":["text"],"outputModalities":["embeddings"],"promptPerMillion":0.005,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"The all-mpnet-base-v2 embedding model encodes sentences and short paragraphs into a 768-dimensional dense vector space, providing high-fidelity semantic embeddings well suited for tasks like information retrieval, clustering, similarity scoring, and...","huggingFaceId":"sentence-transformers/all-mpnet-base-v2","routes":[{"provider":"OpenRouter","modelId":"sentence-transformers/all-mpnet-base-v2","url":"https://openrouter.ai/sentence-transformers/all-mpnet-base-v2","free":false,"batch":false}],"benchmarks":[{"benchmark":"BRIGHT · BrightAopsRetrieval default standard","score":5.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-03-26","scope":"Registered model evaluation · Obtained using MTEB v2.10.12"},{"benchmark":"BRIGHT · BrightAopsRetrieval","score":5.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-03-26","scope":"Registered model evaluation · Obtained using MTEB v2.10.12"},{"benchmark":"BRIGHT · BrightBiologyLongRetrieval default long","score":25.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-03-26","scope":"Registered model evaluation · Obtained using MTEB v2.10.12"},{"benchmark":"BRIGHT · BrightBiologyLongRetrieval","score":25.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-03-26","scope":"Registered model evaluation · Obtained using MTEB v2.10.12"},{"benchmark":"BRIGHT · BrightBiologyRetrieval default standard","score":15.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-03-26","scope":"Registered model evaluation · Obtained using MTEB v2.10.12"},{"benchmark":"BRIGHT · BrightBiologyRetrieval","score":15.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-03-26","scope":"Registered model evaluation · Obtained using MTEB v2.10.12"},{"benchmark":"BRIGHT · BrightEarthScienceLongRetrieval default long","score":34.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-03-26","scope":"Registered model evaluation · Obtained using MTEB v2.10.12"},{"benchmark":"BRIGHT · BrightEarthScienceLongRetrieval","score":34.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-03-26","scope":"Registered model evaluation · Obtained using MTEB v2.10.12"},{"benchmark":"MTEB English retrieval","score":44.5,"unit":"%","sourceName":"MTEB","sourceUrl":"https://leaderboard.mteb.org/","measuredAt":"2026-08-27","scope":"Mean retrieval score · leaderboard rank 121"}],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[{"id":"a8358232-73db-4250-b641-8826009c7e68","title":"AI Autoresearch Strategy Generator","href":"/intel/a8358232-73db-4250-b641-8826009c7e68","hook":"Parallel Claude agents that generate, test, and optimize trading strategies — and won a hackathon doing it. Read it for the parallel-agent search pattern, wh…"},{"id":"d368d9a4-37d9-456a-bc0c-081ae8dd4c28","title":"The SaaS Reckoning","href":"/intel/d368d9a4-37d9-456a-bc0c-081ae8dd4c28","hook":"How stock-based comp has masked the true costs of SaaS, and why AI disruption forces the bill due. A financially literate take for founders and operators who…"},{"id":"49e0adc5-ae09-46c8-8d10-582885877804","title":"Agent Harness Memory Control · article","href":"/intel/49e0adc5-ae09-46c8-8d10-582885877804","hook":"Why the harness — and specifically who controls your agent's memory — is critical infrastructure, and why open harnesses matter. A short, sharp argument for …"}]},{"id":"sentence-transformers/all-minilm-l12-v2-20251117","name":"all-MiniLM-L12-v2","releaseLabel":"MiniLM-L12-v2","generationId":"v2","generationLabel":"V2","variantLabel":"Minilm L12","createdAt":"2025-11-18T02:15:55.000Z","kind":"embedding","contextLength":512,"inputModalities":["text"],"outputModalities":["embeddings"],"promptPerMillion":0.005,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"The all-MiniLM-L12-v2 embedding model maps sentences and short paragraphs into a 384-dimensional dense vector space, producing efficient and high-quality semantic embeddings optimized for tasks such as semantic search, clustering, and...","huggingFaceId":"sentence-transformers/all-MiniLM-L12-v2","routes":[{"provider":"OpenRouter","modelId":"sentence-transformers/all-minilm-l12-v2","url":"https://openrouter.ai/sentence-transformers/all-minilm-l12-v2","free":false,"batch":false}],"benchmarks":[{"benchmark":"arguana · ArguAna default test","score":47.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-03-05","scope":"Registered model evaluation · Obtained using MTEB v1.38.3"},{"benchmark":"arguana · ArguAna","score":47.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-03-05","scope":"Registered model evaluation · Obtained using MTEB v1.38.3"},{"benchmark":"MTEB English retrieval","score":43.7,"unit":"%","sourceName":"MTEB","sourceUrl":"https://leaderboard.mteb.org/","measuredAt":"2026-08-27","scope":"Mean retrieval score · leaderboard rank 159"}],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[{"id":"a8358232-73db-4250-b641-8826009c7e68","title":"AI Autoresearch Strategy Generator","href":"/intel/a8358232-73db-4250-b641-8826009c7e68","hook":"Parallel Claude agents that generate, test, and optimize trading strategies — and won a hackathon doing it. Read it for the parallel-agent search pattern, wh…"},{"id":"d368d9a4-37d9-456a-bc0c-081ae8dd4c28","title":"The SaaS Reckoning","href":"/intel/d368d9a4-37d9-456a-bc0c-081ae8dd4c28","hook":"How stock-based comp has masked the true costs of SaaS, and why AI disruption forces the bill due. A financially literate take for founders and operators who…"},{"id":"49e0adc5-ae09-46c8-8d10-582885877804","title":"Agent Harness Memory Control · article","href":"/intel/49e0adc5-ae09-46c8-8d10-582885877804","hook":"Why the harness — and specifically who controls your agent's memory — is critical infrastructure, and why open harnesses matter. A short, sharp argument for …"}]}]},{"id":"multi","label":"Multi","models":[{"id":"sentence-transformers/multi-qa-mpnet-base-dot-v1-20251117","name":"multi-qa-mpnet-base-dot-v1","releaseLabel":"qa-mpnet-base-dot-v1","generationId":"v1","generationLabel":"V1","variantLabel":"Qa Mpnet Base Dot","createdAt":"2025-11-18T02:02:19.000Z","kind":"embedding","contextLength":512,"inputModalities":["text"],"outputModalities":["embeddings"],"promptPerMillion":0.005,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"The multi-qa-mpnet-base-dot-v1 embedding model transforms sentences and short paragraphs into a 768-dimensional dense vector space, generating high-quality semantic embeddings optimized for question-and-answer retrieval, semantic search, and similarity-scoring across diverse content.","huggingFaceId":"sentence-transformers/multi-qa-mpnet-base-dot-v1","routes":[{"provider":"OpenRouter","modelId":"sentence-transformers/multi-qa-mpnet-base-dot-v1","url":"https://openrouter.ai/sentence-transformers/multi-qa-mpnet-base-dot-v1","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[{"id":"f7f80504-c539-4937-a09e-7ff21d0a67fc","title":"ChatDev 2.0","href":"/intel/f7f80504-c539-4937-a09e-7ff21d0a67fc","hook":"Zero-code multi-agent orchestration: you describe the system in config and it builds and runs the agent team. Worth a look if you are tired of hand-wiring ag…"},{"id":"edaf640d-7072-4122-9cb3-3f8611338b51","title":"LLM Research Papers: The 2025 List","href":"/intel/edaf640d-7072-4122-9cb3-3f8611338b51","hook":"A single, thematically organized reference to 200+ of the most important 2025 LLM papers — grouped by reasoning, RL, and multimodal themes — saves researcher…"},{"id":"0e02d304-1b0c-470b-bb55-1cf6b544d3d1","title":"GPT-5","href":"/intel/0e02d304-1b0c-470b-bb55-1cf6b544d3d1","hook":"GPT-5's default toward autonomous, multi-step action on vague prompts changes how you scope tasks for it — Mollick's hands-on examples show where that proact…"}]}]},{"id":"paraphrase","label":"Paraphrase","models":[{"id":"sentence-transformers/paraphrase-minilm-l6-v2-20251117","name":"paraphrase-MiniLM-L6-v2","releaseLabel":"MiniLM-L6-v2","generationId":"v2","generationLabel":"V2","variantLabel":"Minilm L6","createdAt":"2025-11-18T02:20:54.000Z","kind":"embedding","contextLength":512,"inputModalities":["text"],"outputModalities":["embeddings"],"promptPerMillion":0.005,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"The paraphrase-MiniLM-L6-v2 embedding model converts sentences and short paragraphs into a 384-dimensional dense vector space, producing high-quality semantic embeddings optimized for paraphrase detection, semantic similarity scoring, clustering, and lightweight retrieval...","huggingFaceId":"sentence-transformers/paraphrase-MiniLM-L6-v2","routes":[{"provider":"OpenRouter","modelId":"sentence-transformers/paraphrase-minilm-l6-v2","url":"https://openrouter.ai/sentence-transformers/paraphrase-minilm-l6-v2","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[]}]}]},{"id":"aion-labs","name":"Aion Labs","domain":null,"major":false,"modelCount":4,"series":[{"id":"aion","label":"Aion","models":[{"id":"aion-labs/aion-rp-llama-3.1-8b","name":"Aion-RP 1.0 (8B)","releaseLabel":"RP 1.0 (8B)","generationId":"3.1","generationLabel":"3.1","variantLabel":"8B","createdAt":"2025-02-04T19:18:38.000Z","kind":"language","contextLength":32768,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.7999999999999999,"completionPerMillion":1.5999999999999999,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Aion-RP-Llama-3.1-8B ranks the highest in the character evaluation portion of the RPBench-Auto benchmark, a roleplaying-specific variant of Arena-Hard-Auto, where LLMs evaluate each other’s responses. It is a fine-tuned base model...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"aion-labs/aion-rp-llama-3.1-8b","url":"https://openrouter.ai/aion-labs/aion-rp-llama-3.1-8b","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"aion-labs/aion-2.0-20260223","name":"Aion-2.0","releaseLabel":"2.0","generationId":"2.0","generationLabel":"2.0","variantLabel":"Base","createdAt":"2026-02-23T21:15:06.000Z","kind":"language","contextLength":131072,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.7999999999999999,"completionPerMillion":1.5999999999999999,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Aion-2.0 is a variant of DeepSeek V3.2 optimized for immersive roleplaying and storytelling. It is particularly strong at introducing tension, crises, and conflict into stories, making narratives feel more engaging....","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"aion-labs/aion-2.0","url":"https://openrouter.ai/aion-labs/aion-2.0","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"aion-labs/aion-3.0-20260707","name":"Aion-3.0","releaseLabel":"3.0","generationId":"3.0","generationLabel":"3.0","variantLabel":"Base","createdAt":"2026-07-07T16:51:35.000Z","kind":"language","contextLength":131072,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":3,"completionPerMillion":6,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Aion-3.0 is a multi-model roleplaying and storytelling system from AionLabs, built on the GLM family of models. It uses a collaborative generation process in which multiple specialized models each contribute...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"aion-labs/aion-3.0","url":"https://openrouter.ai/aion-labs/aion-3.0","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"aion-labs/aion-3.0-mini-20260707","name":"Aion-3.0-Mini","releaseLabel":"3.0-Mini","generationId":"3.0","generationLabel":"3.0","variantLabel":"Mini","createdAt":"2026-07-07T16:51:36.000Z","kind":"language","contextLength":131072,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.7,"completionPerMillion":1.4,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Aion-3.0 Mini is a multi-model roleplaying and storytelling system from AionLabs, built on the DeepSeek family of models. It uses a collaborative generation process in which multiple specialized models each...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"aion-labs/aion-3.0-mini","url":"https://openrouter.ai/aion-labs/aion-3.0-mini","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]}]},{"id":"nousresearch","name":"Nous Research","domain":"nousresearch.com","major":false,"modelCount":4,"series":[{"id":"hermes","label":"Hermes","models":[{"id":"nousresearch/hermes-3-llama-3.1-405b","name":"Hermes 3 405B Instruct","releaseLabel":"3 405B Instruct","generationId":"3","generationLabel":"3","variantLabel":"Llama 3.1 405B","createdAt":"2024-08-16T00:00:00.000Z","kind":"language","contextLength":131072,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":1,"completionPerMillion":1,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Hermes 3 is a generalist language model with many improvements over Hermes 2, including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","huggingFaceId":"NousResearch/Hermes-3-Llama-3.1-405B","routes":[{"provider":"OpenRouter","modelId":"nousresearch/hermes-3-llama-3.1-405b","url":"https://openrouter.ai/nousresearch/hermes-3-llama-3.1-405b","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"nousresearch/hermes-3-llama-3.1-70b","name":"Hermes 3 70B Instruct","releaseLabel":"3 70B Instruct","generationId":"3","generationLabel":"3","variantLabel":"Llama 3.1 70B","createdAt":"2024-08-18T00:00:00.000Z","kind":"language","contextLength":131072,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.7,"completionPerMillion":0.7,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Hermes 3 is a generalist language model with many improvements over [Hermes 2](/models/nousresearch/nous-hermes-2-mistral-7b-dpo), including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","huggingFaceId":"NousResearch/Hermes-3-Llama-3.1-70B","routes":[{"provider":"OpenRouter","modelId":"nousresearch/hermes-3-llama-3.1-70b","url":"https://openrouter.ai/nousresearch/hermes-3-llama-3.1-70b","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"nousresearch/hermes-4-405b","name":"Hermes 4 405B","releaseLabel":"4 405B","generationId":"4","generationLabel":"4","variantLabel":"405B","createdAt":"2025-08-26T19:11:03.000Z","kind":"language","contextLength":131072,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":1,"completionPerMillion":3,"imageOutput":null,"supportsTools":false,"supportsReasoning":true,"description":"Hermes 4 is a large-scale reasoning model built on Meta-Llama-3.1-405B and released by Nous Research. It introduces a hybrid reasoning mode, where the model can choose to deliberate internally with...","huggingFaceId":"NousResearch/Hermes-4-405B","routes":[{"provider":"OpenRouter","modelId":"nousresearch/hermes-4-405b","url":"https://openrouter.ai/nousresearch/hermes-4-405b","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"nousresearch/hermes-4-70b","name":"Hermes 4 70B","releaseLabel":"4 70B","generationId":"4","generationLabel":"4","variantLabel":"70B","createdAt":"2025-08-26T19:23:02.000Z","kind":"language","contextLength":131072,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.13,"completionPerMillion":0.39999999999999997,"imageOutput":null,"supportsTools":false,"supportsReasoning":true,"description":"Hermes 4 70B is a hybrid reasoning model from Nous Research, built on Meta-Llama-3.1-70B. It introduces the same hybrid mode as the larger 405B release, allowing the model to either...","huggingFaceId":"NousResearch/Hermes-4-70B","routes":[{"provider":"OpenRouter","modelId":"nousresearch/hermes-4-70b","url":"https://openrouter.ai/nousresearch/hermes-4-70b","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]}]},{"id":"sourceful","name":"Sourceful","domain":null,"major":false,"modelCount":4,"series":[{"id":"riverflow","label":"Riverflow","models":[{"id":"sourceful/riverflow-v2-fast-20260130","name":"Riverflow V2 Fast","releaseLabel":"V2 Fast","generationId":"v2","generationLabel":"V2","variantLabel":"Fast","createdAt":"2026-02-02T16:57:03.000Z","kind":"image","contextLength":8192,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":4.79041916167665,"supportsTools":false,"supportsReasoning":false,"description":"Riverflow V2 Fast is the fastest variant of Sourceful's Riverflow 2.0 lineup, best for production deployments and latency-critical workflows. The Riverflow 2.0 series represents SOTA performance on image generation and...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"sourceful/riverflow-v2-fast","url":"https://openrouter.ai/sourceful/riverflow-v2-fast","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[]},{"id":"sourceful/riverflow-v2-pro-20260130","name":"Riverflow V2 Pro","releaseLabel":"V2 Pro","generationId":"v2","generationLabel":"V2","variantLabel":"Pro","createdAt":"2026-02-02T16:57:07.000Z","kind":"image","contextLength":8192,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":35.9281437125749,"supportsTools":false,"supportsReasoning":false,"description":"Riverflow V2 Pro is the most powerful variant of Sourceful's Riverflow 2.0 lineup, best for top-tier control and perfect text rendering. The Riverflow 2.0 series represents SOTA performance on image...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"sourceful/riverflow-v2-pro","url":"https://openrouter.ai/sourceful/riverflow-v2-pro","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[]},{"id":"sourceful/riverflow-v2.5-fast-20260605","name":"Riverflow V2.5 Fast","releaseLabel":"V2.5 Fast","generationId":"v2.5","generationLabel":"V2.5","variantLabel":"Fast","createdAt":"2026-06-04T14:56:23.000Z","kind":"image","contextLength":32768,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":4.55089820359281,"supportsTools":false,"supportsReasoning":true,"description":"Riverflow V2.5 Fast is the speed-optimized variant of Sourceful's Riverflow 2.5 lineup, best for production deployments and latency-critical workflows. The Riverflow 2.5 series is a unified text-to-image and image-to-image family...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"sourceful/riverflow-v2.5-fast","url":"https://openrouter.ai/sourceful/riverflow-v2.5-fast","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[]},{"id":"sourceful/riverflow-v2.5-pro-20260605","name":"Riverflow V2.5 Pro","releaseLabel":"V2.5 Pro","generationId":"v2.5","generationLabel":"V2.5","variantLabel":"Pro","createdAt":"2026-06-04T14:56:31.000Z","kind":"image","contextLength":32768,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":31.1377245508982,"supportsTools":false,"supportsReasoning":true,"description":"Riverflow V2.5 Pro is the most powerful variant of Sourceful's Riverflow 2.5 lineup, best for top-tier control and quality-sensitive outputs. The Riverflow 2.5 series is a unified text-to-image and image-to-image...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"sourceful/riverflow-v2.5-pro","url":"https://openrouter.ai/sourceful/riverflow-v2.5-pro","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[]}]}]},{"id":"thedrummer","name":"Thedrummer","domain":null,"major":false,"modelCount":4,"series":[{"id":"cydonia","label":"Cydonia","models":[{"id":"thedrummer/cydonia-24b-v4.1","name":"Cydonia 24B V4.1","releaseLabel":"24B V4.1","generationId":"24b","generationLabel":"24B","variantLabel":"V4.1","createdAt":"2025-09-27T00:11:18.000Z","kind":"language","contextLength":131072,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.3,"completionPerMillion":0.5,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Uncensored and creative writing model based on Mistral Small 3.2 24B with good recall, prompt adherence, and intelligence.","huggingFaceId":"thedrummer/cydonia-24b-v4.1","routes":[{"provider":"OpenRouter","modelId":"thedrummer/cydonia-24b-v4.1","url":"https://openrouter.ai/thedrummer/cydonia-24b-v4.1","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]},{"id":"rocinante","label":"Rocinante","models":[{"id":"thedrummer/rocinante-12b","name":"Rocinante 12B","releaseLabel":"12B","generationId":"12b","generationLabel":"12B","variantLabel":"Base","createdAt":"2024-09-30T00:00:00.000Z","kind":"language","contextLength":65536,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.25,"completionPerMillion":0.5,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Rocinante 12B is designed for engaging storytelling and rich prose. Early testers have reported: - Expanded vocabulary with unique and expressive word choices - Enhanced creativity for vivid narratives -...","huggingFaceId":"TheDrummer/Rocinante-12B-v1.1","routes":[{"provider":"OpenRouter","modelId":"thedrummer/rocinante-12b","url":"https://openrouter.ai/thedrummer/rocinante-12b","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]},{"id":"skyfall","label":"Skyfall","models":[{"id":"thedrummer/skyfall-36b-v2","name":"Skyfall 36B V2","releaseLabel":"36B V2","generationId":"36b","generationLabel":"36B","variantLabel":"V2","createdAt":"2025-03-10T19:56:06.000Z","kind":"language","contextLength":32768,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.55,"completionPerMillion":0.7999999999999999,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Skyfall 36B v2 is an enhanced iteration of Mistral Small 2501, specifically fine-tuned for improved creativity, nuanced writing, role-playing, and coherent storytelling.","huggingFaceId":"TheDrummer/Skyfall-36B-v2","routes":[{"provider":"OpenRouter","modelId":"thedrummer/skyfall-36b-v2","url":"https://openrouter.ai/thedrummer/skyfall-36b-v2","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]},{"id":"unslopnemo","label":"Unslopnemo","models":[{"id":"thedrummer/unslopnemo-12b","name":"UnslopNemo 12B","releaseLabel":"12B","generationId":"12b","generationLabel":"12B","variantLabel":"Base","createdAt":"2024-11-08T22:04:08.000Z","kind":"language","contextLength":1024000,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.39999999999999997,"completionPerMillion":0.39999999999999997,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"UnslopNemo v4.1 is the latest addition from the creator of Rocinante, designed for adventure writing and role-play scenarios.","huggingFaceId":"TheDrummer/UnslopNemo-12B-v4.1","routes":[{"provider":"OpenRouter","modelId":"thedrummer/unslopnemo-12b","url":"https://openrouter.ai/thedrummer/unslopnemo-12b","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]}]},{"id":"baai","name":"BAAI","domain":"baai.ac.cn","major":false,"modelCount":3,"series":[{"id":"bge","label":"BGE","models":[{"id":"baai/bge-m3-20251117","name":"bge-m3","releaseLabel":"m3","generationId":"releases","generationLabel":"Releases","variantLabel":"M3","createdAt":"2025-11-18T00:06:12.000Z","kind":"embedding","contextLength":8194,"inputModalities":["text"],"outputModalities":["embeddings"],"promptPerMillion":0.01,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"The bge-m3 embedding model encodes sentences, paragraphs, and long documents into a 1024-dimensional dense vector space, delivering high-quality semantic embeddings optimized for multilingual retrieval, semantic search, and large-context applications.","huggingFaceId":"BAAI/bge-m3","routes":[{"provider":"OpenRouter","modelId":"baai/bge-m3","url":"https://openrouter.ai/baai/bge-m3","free":false,"batch":false}],"benchmarks":[{"benchmark":"BRIGHT · BrightAopsRetrieval default standard","score":4.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-04-01","scope":"Registered model evaluation · Obtained using MTEB v2.10.12"},{"benchmark":"BRIGHT · BrightAopsRetrieval","score":4.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-04-01","scope":"Registered model evaluation · Obtained using MTEB v2.10.12"},{"benchmark":"BRIGHT · BrightBiologyLongRetrieval default long","score":14.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-04-01","scope":"Registered model evaluation · Obtained using MTEB v2.10.12"},{"benchmark":"BRIGHT · BrightBiologyLongRetrieval","score":14.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-04-01","scope":"Registered model evaluation · Obtained using MTEB v2.10.12"},{"benchmark":"BRIGHT · BrightBiologyRetrieval default standard","score":9.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-04-01","scope":"Registered model evaluation · Obtained using MTEB v2.10.12"},{"benchmark":"BRIGHT · BrightBiologyRetrieval","score":9.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-04-01","scope":"Registered model evaluation · Obtained using MTEB v2.10.12"},{"benchmark":"BRIGHT · BrightEarthScienceLongRetrieval default long","score":20.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-04-01","scope":"Registered model evaluation · Obtained using MTEB v2.10.12"},{"benchmark":"BRIGHT · BrightEarthScienceLongRetrieval","score":20.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-04-01","scope":"Registered model evaluation · Obtained using MTEB v2.10.12"}],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[]},{"id":"baai/bge-large-en-v1.5-20251117","name":"bge-large-en-v1.5","releaseLabel":"large-en-v1.5","generationId":"v1.5","generationLabel":"V1.5","variantLabel":"Large En","createdAt":"2025-11-18T01:58:07.000Z","kind":"embedding","contextLength":512,"inputModalities":["text"],"outputModalities":["embeddings"],"promptPerMillion":0.01,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"The bge-large-en-v1.5 embedding model maps English sentences, paragraphs, and documents into a 1024-dimensional dense vector space, delivering high-fidelity semantic embeddings optimized for semantic search, document retrieval, and downstream NLP tasks...","huggingFaceId":"BAAI/bge-large-en-v1.5","routes":[{"provider":"OpenRouter","modelId":"baai/bge-large-en-v1.5","url":"https://openrouter.ai/baai/bge-large-en-v1.5","free":false,"batch":false}],"benchmarks":[{"benchmark":"BRIGHT · BrightAopsRetrieval default standard","score":6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-03-26","scope":"Registered model evaluation · Obtained using MTEB v2.10.12"},{"benchmark":"BRIGHT · BrightAopsRetrieval","score":6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-03-26","scope":"Registered model evaluation · Obtained using MTEB v2.10.12"},{"benchmark":"BRIGHT · BrightBiologyLongRetrieval default long","score":16.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-03-26","scope":"Registered model evaluation · Obtained using MTEB v2.10.12"},{"benchmark":"BRIGHT · BrightBiologyLongRetrieval","score":16.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-03-26","scope":"Registered model evaluation · Obtained using MTEB v2.10.12"},{"benchmark":"BRIGHT · BrightBiologyRetrieval default standard","score":11.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-03-26","scope":"Registered model evaluation · Obtained using MTEB v2.10.12"},{"benchmark":"BRIGHT · BrightBiologyRetrieval","score":11.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-03-26","scope":"Registered model evaluation · Obtained using MTEB v2.10.12"},{"benchmark":"BRIGHT · BrightEarthScienceLongRetrieval default long","score":27.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-03-26","scope":"Registered model evaluation · Obtained using MTEB v2.10.12"},{"benchmark":"BRIGHT · BrightEarthScienceLongRetrieval","score":27.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-03-26","scope":"Registered model evaluation · Obtained using MTEB v2.10.12"},{"benchmark":"MTEB English retrieval","score":55.4,"unit":"%","sourceName":"MTEB","sourceUrl":"https://leaderboard.mteb.org/","measuredAt":"2026-08-27","scope":"Mean retrieval score · leaderboard rank 64"}],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[]},{"id":"baai/bge-base-en-v1.5-20251117","name":"bge-base-en-v1.5","releaseLabel":"base-en-v1.5","generationId":"v1.5","generationLabel":"V1.5","variantLabel":"Base En","createdAt":"2025-11-18T02:10:37.000Z","kind":"embedding","contextLength":512,"inputModalities":["text"],"outputModalities":["embeddings"],"promptPerMillion":0.005,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"The bge-base-en-v1.5 embedding model converts English sentences and paragraphs into 768-dimensional dense vectors, delivering efficient, high-quality semantic embeddings optimized for retrieval, semantic search, and document-matching workflows. This version (v1.5) features...","huggingFaceId":"BAAI/bge-base-en-v1.5","routes":[{"provider":"OpenRouter","modelId":"baai/bge-base-en-v1.5","url":"https://openrouter.ai/baai/bge-base-en-v1.5","free":false,"batch":false}],"benchmarks":[{"benchmark":"MTEB English retrieval","score":54.8,"unit":"%","sourceName":"MTEB","sourceUrl":"https://leaderboard.mteb.org/","measuredAt":"2026-08-27","scope":"Mean retrieval score · leaderboard rank 72"}],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[]}]}]},{"id":"deepgram","name":"Deepgram","domain":"deepgram.com","major":false,"modelCount":3,"series":[{"id":"aura","label":"Aura","models":[{"id":"deepgram/aura-2-20260716","name":"Aura-2","releaseLabel":"2","generationId":"2","generationLabel":"2","variantLabel":"Base","createdAt":"2026-07-16T21:26:07.000Z","kind":"audio","contextLength":0,"inputModalities":["text"],"outputModalities":["speech"],"promptPerMillion":30,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Aura-2 is a multilingual text-to-speech model from Deepgram. It supports Deepgram’s canonical Aura-2 voice catalog for speech synthesis across multiple languages.","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"deepgram/aura-2","url":"https://openrouter.ai/deepgram/aura-2","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[]}]},{"id":"flux","label":"FLUX","models":[{"id":"deepgram/flux-tts-20260812","name":"Flux TTS","releaseLabel":"TTS","generationId":"releases","generationLabel":"Releases","variantLabel":"TTS","createdAt":"2026-08-12T22:48:08.000Z","kind":"audio","contextLength":0,"inputModalities":["text"],"outputModalities":["speech"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Flux TTS is a text-to-speech model from Deepgram. It is suited for natural, expressive English speech synthesis across Deepgram's Flux voice catalog.","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"deepgram/flux-tts:free","url":"https://openrouter.ai/deepgram/flux-tts:free","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[]}]},{"id":"nova","label":"Nova","models":[{"id":"deepgram/nova-3-20260714","name":"Nova-3","releaseLabel":"3","generationId":"3","generationLabel":"3","variantLabel":"Base","createdAt":"2026-07-15T00:36:01.000Z","kind":"audio","contextLength":0,"inputModalities":["audio"],"outputModalities":["transcription"],"promptPerMillion":4300,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Deepgram Nova-3 general-purpose speech-to-text model with monolingual and multilingual transcription support.","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"deepgram/nova-3","url":"https://openrouter.ai/deepgram/nova-3","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[]}]}]},{"id":"intfloat","name":"Intfloat","domain":null,"major":false,"modelCount":3,"series":[{"id":"e5","label":"E5","models":[{"id":"intfloat/e5-base-v2-20251117","name":"E5-Base-v2","releaseLabel":"Base-v2","generationId":"v2","generationLabel":"V2","variantLabel":"Base","createdAt":"2025-11-18T02:33:12.000Z","kind":"embedding","contextLength":512,"inputModalities":["text"],"outputModalities":["embeddings"],"promptPerMillion":0.005,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"The e5-base-v2 embedding model encodes English sentences and paragraphs into a 768-dimensional dense vector space, producing efficient and high-quality semantic embeddings optimized for tasks such as semantic search, similarity scoring,...","huggingFaceId":"intfloat/e5-base-v2","routes":[{"provider":"OpenRouter","modelId":"intfloat/e5-base-v2","url":"https://openrouter.ai/intfloat/e5-base-v2","free":false,"batch":false}],"benchmarks":[{"benchmark":"MTEB English retrieval","score":49.7,"unit":"%","sourceName":"MTEB","sourceUrl":"https://leaderboard.mteb.org/","measuredAt":"2026-08-27","scope":"Mean retrieval score · leaderboard rank 107"}],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[]},{"id":"intfloat/e5-large-v2-20251117","name":"E5-Large-v2","releaseLabel":"Large-v2","generationId":"v2","generationLabel":"V2","variantLabel":"Large","createdAt":"2025-11-18T02:37:12.000Z","kind":"embedding","contextLength":512,"inputModalities":["text"],"outputModalities":["embeddings"],"promptPerMillion":0.01,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"The e5-large-v2 embedding model maps English sentences, paragraphs, and documents into a 1024-dimensional dense vector space, delivering high-accuracy semantic embeddings optimized for retrieval, semantic search, reranking, and similarity-scoring tasks.","huggingFaceId":"intfloat/e5-large-v2","routes":[{"provider":"OpenRouter","modelId":"intfloat/e5-large-v2","url":"https://openrouter.ai/intfloat/e5-large-v2","free":false,"batch":false}],"benchmarks":[{"benchmark":"MTEB English retrieval","score":49.3,"unit":"%","sourceName":"MTEB","sourceUrl":"https://leaderboard.mteb.org/","measuredAt":"2026-08-27","scope":"Mean retrieval score · leaderboard rank 101"}],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[]}]},{"id":"multilingual","label":"Multilingual","models":[{"id":"intfloat/multilingual-e5-large-20251117","name":"Multilingual-E5-Large","releaseLabel":"E5-Large","generationId":"releases","generationLabel":"Releases","variantLabel":"E5 Large","createdAt":"2025-11-18T02:30:47.000Z","kind":"embedding","contextLength":512,"inputModalities":["text"],"outputModalities":["embeddings"],"promptPerMillion":0.01,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"The multilingual-e5-large embedding model encodes sentences, paragraphs, and documents across over 90 languages into a 1024-dimensional dense vector space, delivering robust semantic embeddings optimized for multilingual retrieval, cross-language similarity, and...","huggingFaceId":"intfloat/multilingual-e5-large","routes":[{"provider":"OpenRouter","modelId":"intfloat/multilingual-e5-large","url":"https://openrouter.ai/intfloat/multilingual-e5-large","free":false,"batch":false}],"benchmarks":[{"benchmark":"arguana · ArguAna default test","score":54.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-04-02","scope":"Registered model evaluation · Obtained using MTEB v1.36.26"},{"benchmark":"arguana · ArguAna","score":54.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-04-02","scope":"Registered model evaluation · Obtained using MTEB v1.36.26"},{"benchmark":"BRIGHT · BrightAopsRetrieval default standard","score":7.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-03-31","scope":"Registered model evaluation · Obtained using MTEB v2.10.12"},{"benchmark":"BRIGHT · BrightAopsRetrieval","score":7.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-03-31","scope":"Registered model evaluation · Obtained using MTEB v2.10.12"},{"benchmark":"BRIGHT · BrightBiologyLongRetrieval default long","score":1.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-03-31","scope":"Registered model evaluation · Obtained using MTEB v2.10.12"},{"benchmark":"BRIGHT · BrightBiologyLongRetrieval","score":1.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-03-31","scope":"Registered model evaluation · Obtained using MTEB v2.10.12"},{"benchmark":"BRIGHT · BrightBiologyRetrieval default standard","score":1.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-03-31","scope":"Registered model evaluation · Obtained using MTEB v2.10.12"},{"benchmark":"BRIGHT · BrightBiologyRetrieval","score":1.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://github.com/embeddings-benchmark/mteb/","measuredAt":"2026-03-31","scope":"Registered model evaluation · Obtained using MTEB v2.10.12"},{"benchmark":"MTEB English retrieval","score":51.5,"unit":"%","sourceName":"MTEB","sourceUrl":"https://leaderboard.mteb.org/","measuredAt":"2026-08-27","scope":"Mean retrieval score · leaderboard rank 122"}],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[]}]}]},{"id":"kwaivgi","name":"Kling AI","domain":"klingai.com","major":false,"modelCount":3,"series":[{"id":"kling","label":"Kling","models":[{"id":"kwaivgi/kling-video-o1-20260420","name":"Video O1","releaseLabel":"Video O1","generationId":"releases","generationLabel":"Releases","variantLabel":"Video O1","createdAt":"2026-04-20T17:06:17.000Z","kind":"video","contextLength":0,"inputModalities":["text","image"],"outputModalities":["video"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Kling Video O1 is a video generation model from Kuaishou. It supports text and image inputs with video output, enabling text-to-video and image-to-video workflows. It is suited for cinematic content...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"kwaivgi/kling-video-o1","url":"https://openrouter.ai/kwaivgi/kling-video-o1","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Video Arena","url":"https://artificialanalysis.ai/video/leaderboard/text-to-video","note":"Blind preference ranking for video generation"}],"intel":[{"id":"0e7e9219-5fd3-4c02-9bac-305501b53419","title":"We ran Thinking Machines Inkling through our private coding-agent bench. Same ha","href":"/intel/0e7e9219-5fd3-4c02-9bac-305501b53419","hook":"Independent benchmark data on Thinking Machines' Inkling across a 13-bug repair harness — a reality check on frontier coding-agent claims."}]},{"id":"kwaivgi/kling-v3.0-std-20260429","name":"Video v3.0 Standard","releaseLabel":"Video v3.0 Standard","generationId":"v3.0","generationLabel":"V3.0","variantLabel":"Std","createdAt":"2026-04-29T20:56:45.000Z","kind":"video","contextLength":0,"inputModalities":["text","image"],"outputModalities":["video"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Kling v3.0 Standard is a video generation model from Kuaishou. It supports text-to-video and image-to-video workflows, with first-frame and last-frame control for guided scene composition. Clips range from 3 to...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"kwaivgi/kling-v3.0-std","url":"https://openrouter.ai/kwaivgi/kling-v3.0-std","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Video Arena","url":"https://artificialanalysis.ai/video/leaderboard/text-to-video","note":"Blind preference ranking for video generation"}],"intel":[{"id":"0e7e9219-5fd3-4c02-9bac-305501b53419","title":"We ran Thinking Machines Inkling through our private coding-agent bench. Same ha","href":"/intel/0e7e9219-5fd3-4c02-9bac-305501b53419","hook":"Independent benchmark data on Thinking Machines' Inkling across a 13-bug repair harness — a reality check on frontier coding-agent claims."}]},{"id":"kwaivgi/kling-v3.0-pro-20260429","name":"Video v3.0 Pro","releaseLabel":"Video v3.0 Pro","generationId":"v3.0","generationLabel":"V3.0","variantLabel":"Pro","createdAt":"2026-04-29T20:56:46.000Z","kind":"video","contextLength":0,"inputModalities":["text","image"],"outputModalities":["video"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Kling v3.0 Pro is Kuaishou's premium video generation model, offering higher visual quality than the Standard tier. It supports text-to-video and image-to-video workflows, with first-frame and last-frame control for precise...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"kwaivgi/kling-v3.0-pro","url":"https://openrouter.ai/kwaivgi/kling-v3.0-pro","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Video Arena","url":"https://artificialanalysis.ai/video/leaderboard/text-to-video","note":"Blind preference ranking for video generation"}],"intel":[{"id":"0e7e9219-5fd3-4c02-9bac-305501b53419","title":"We ran Thinking Machines Inkling through our private coding-agent bench. Same ha","href":"/intel/0e7e9219-5fd3-4c02-9bac-305501b53419","hook":"Independent benchmark data on Thinking Machines' Inkling across a 13-bug repair harness — a reality check on frontier coding-agent claims."}]}]}]},{"id":"krea","name":"Krea","domain":"krea.ai","major":false,"modelCount":3,"series":[{"id":"krea","label":"Krea","models":[{"id":"krea/krea-2-medium-turbo-20260720","name":"Krea 2 Medium Turbo","releaseLabel":"2 Medium Turbo","generationId":"2","generationLabel":"2","variantLabel":"Medium Turbo","createdAt":"2026-07-20T19:15:23.000Z","kind":"image","contextLength":65536,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":3.59281437125749,"supportsTools":false,"supportsReasoning":false,"description":"Krea 2 Medium Turbo is a distilled, speed-focused variant of Krea 2 Medium from Krea. It is designed for rapid iteration and graphic design exploration where fast generation is the...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"krea/krea-2-medium-turbo","url":"https://openrouter.ai/krea/krea-2-medium-turbo","free":true,"batch":false}],"benchmarks":[{"benchmark":"AA Image Arena Elo","score":1215,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/image/leaderboard/text-to-image","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[]},{"id":"krea/krea-2-medium-20260720","name":"Krea 2 Medium","releaseLabel":"2 Medium","generationId":"2","generationLabel":"2","variantLabel":"Medium","createdAt":"2026-07-20T19:15:28.000Z","kind":"image","contextLength":65536,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":7.185628742514971,"supportsTools":false,"supportsReasoning":false,"description":"Krea 2 Medium is Krea's balanced, cost-efficient image generation model and a practical starting point for a broad range of use cases. Its extensive post-training supports stable, consistent generations, with...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"krea/krea-2-medium","url":"https://openrouter.ai/krea/krea-2-medium","free":true,"batch":false}],"benchmarks":[{"benchmark":"AA Image Arena Elo","score":1215,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/image/leaderboard/text-to-image","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[]},{"id":"krea/krea-2-large-20260720","name":"Krea 2 Large","releaseLabel":"2 Large","generationId":"2","generationLabel":"2","variantLabel":"Large","createdAt":"2026-07-20T19:15:31.000Z","kind":"image","contextLength":65536,"inputModalities":["text","image"],"outputModalities":["image"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":14.3712574850299,"supportsTools":false,"supportsReasoning":false,"description":"Krea 2 Large is Krea's high-capability image generation model, more than twice the size of Krea 2 Medium. Its lighter post-training gives images a rawer, more textured, and flexible character,...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"krea/krea-2-large","url":"https://openrouter.ai/krea/krea-2-large","free":true,"batch":false}],"benchmarks":[{"benchmark":"AA Image Arena Elo","score":1221,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/image/leaderboard/text-to-image","measuredAt":"2026-08-27","scope":"Current blind human-preference arena rating"}],"benchmarkLinks":[{"label":"Artificial Analysis Image Arena","url":"https://artificialanalysis.ai/image/leaderboard/text-to-image","note":"Blind preference ranking for image generation"}],"intel":[]}]}]},{"id":"kwaipilot","name":"Kwaipilot","domain":null,"major":false,"modelCount":3,"series":[{"id":"kat","label":"KAT","models":[{"id":"kwaipilot/kat-coder-pro-v2-20260327","name":"KAT-Coder-Pro V2","releaseLabel":"Coder-Pro V2","generationId":"v2","generationLabel":"V2","variantLabel":"Coder Pro","createdAt":"2026-03-27T22:08:30.000Z","kind":"language","contextLength":262144,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.3,"completionPerMillion":1.2,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"KAT-Coder-Pro V2 is the latest high-performance model in KwaiKAT’s KAT-Coder series, designed for complex enterprise-grade software engineering and SaaS integration. It builds on the agentic coding strengths of earlier versions,...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"kwaipilot/kat-coder-pro-v2","url":"https://openrouter.ai/kwaipilot/kat-coder-pro-v2","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"a8358232-73db-4250-b641-8826009c7e68","title":"AI Autoresearch Strategy Generator","href":"/intel/a8358232-73db-4250-b641-8826009c7e68","hook":"Parallel Claude agents that generate, test, and optimize trading strategies — and won a hackathon doing it. Read it for the parallel-agent search pattern, wh…"}]},{"id":"kwaipilot/kat-coder-pro-v2.5-20260710","name":"KAT-Coder-Pro V2.5","releaseLabel":"Coder-Pro V2.5","generationId":"v2.5","generationLabel":"V2.5","variantLabel":"Coder Pro","createdAt":"2026-07-10T20:16:29.000Z","kind":"language","contextLength":256000,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.74,"completionPerMillion":2.96,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"KAT-Coder-Pro V2.5 is a flagship-level Agentic Coding model that can directly hand over an entire issue or an entire business workflow to it, allowing it to autonomously locate and make...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"kwaipilot/kat-coder-pro-v2.5","url":"https://openrouter.ai/kwaipilot/kat-coder-pro-v2.5","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"a8358232-73db-4250-b641-8826009c7e68","title":"AI Autoresearch Strategy Generator","href":"/intel/a8358232-73db-4250-b641-8826009c7e68","hook":"Parallel Claude agents that generate, test, and optimize trading strategies — and won a hackathon doing it. Read it for the parallel-agent search pattern, wh…"}]},{"id":"kwaipilot/kat-coder-air-v2.5-20260710","name":"KAT-Coder-Air V2.5","releaseLabel":"Coder-Air V2.5","generationId":"v2.5","generationLabel":"V2.5","variantLabel":"Coder Air","createdAt":"2026-07-10T20:16:30.000Z","kind":"language","contextLength":256000,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.15,"completionPerMillion":0.6,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"KAT-Coder-Air V2.5 is a flagship-level Agentic Coding model that can directly hand over an entire issue or an entire business workflow to it, allowing it to autonomously locate and make...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"kwaipilot/kat-coder-air-v2.5","url":"https://openrouter.ai/kwaipilot/kat-coder-air-v2.5","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"a8358232-73db-4250-b641-8826009c7e68","title":"AI Autoresearch Strategy Generator","href":"/intel/a8358232-73db-4250-b641-8826009c7e68","hook":"Parallel Claude agents that generate, test, and optimize trading strategies — and won a hackathon doing it. Read it for the parallel-agent search pattern, wh…"}]}]}]},{"id":"sao10k","name":"Sao10k","domain":null,"major":false,"modelCount":3,"series":[{"id":"l3","label":"L3","models":[{"id":"sao10k/l3-lunaris-8b","name":"Llama 3 8B Lunaris","releaseLabel":"Llama 3 8B Lunaris","generationId":"8b","generationLabel":"8B","variantLabel":"Llama 3 Lunaris","createdAt":"2024-08-13T00:00:00.000Z","kind":"language","contextLength":8192,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.04,"completionPerMillion":0.049999999999999996,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Lunaris 8B is a versatile generalist and roleplaying model based on Llama 3. It's a strategic merge of multiple models, designed to balance creativity with improved logic and general knowledge....","huggingFaceId":"Sao10K/L3-8B-Lunaris-v1","routes":[{"provider":"OpenRouter","modelId":"sao10k/l3-lunaris-8b","url":"https://openrouter.ai/sao10k/l3-lunaris-8b","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]},{"id":"l3.1","label":"L3.1","models":[{"id":"sao10k/l3.1-euryale-70b","name":"Llama 3.1 Euryale 70B v2.2","releaseLabel":"Llama 3.1 Euryale 70B v2.2","generationId":"70b","generationLabel":"70B","variantLabel":"Llama 3.1 Euryale V2.2","createdAt":"2024-08-28T00:00:00.000Z","kind":"language","contextLength":131072,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.85,"completionPerMillion":0.85,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Euryale L3.1 70B v2.2 is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.1](/models/sao10k/l3-euryale-70b).","huggingFaceId":"Sao10K/L3.1-70B-Euryale-v2.2","routes":[{"provider":"OpenRouter","modelId":"sao10k/l3.1-euryale-70b","url":"https://openrouter.ai/sao10k/l3.1-euryale-70b","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]},{"id":"l3.3","label":"L3.3","models":[{"id":"sao10k/l3.3-euryale-70b-v2.3","name":"Llama 3.3 Euryale 70B","releaseLabel":"Llama 3.3 Euryale 70B","generationId":"70b","generationLabel":"70B","variantLabel":"Llama 3.3 Euryale","createdAt":"2024-12-18T15:32:08.000Z","kind":"language","contextLength":131072,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.65,"completionPerMillion":0.75,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Euryale L3.3 70B is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.2](/models/sao10k/l3-euryale-70b).","huggingFaceId":"Sao10K/L3.3-70B-Euryale-v2.3","routes":[{"provider":"OpenRouter","modelId":"sao10k/l3.3-euryale-70b","url":"https://openrouter.ai/sao10k/l3.3-euryale-70b","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]}]},{"id":"xiaomi","name":"Xiaomi","domain":"mi.com","major":false,"modelCount":3,"series":[{"id":"mimo","label":"Mimo","models":[{"id":"xiaomi/mimo-v2-flash","name":"MiMo-V2-Flash","releaseLabel":"V2-Flash","generationId":"v2","generationLabel":"V2","variantLabel":"Flash","createdAt":"2025-12-16T00:00:00.000Z","kind":"language","contextLength":262144,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":null,"completionPerMillion":null,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Xiaomi's open-weight reasoning, coding, and agentic foundation model.","huggingFaceId":"XiaomiMiMo/MiMo-V2-Flash","routes":[],"benchmarks":[{"benchmark":"AA Intelligence Index","score":25.1,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-flash","measuredAt":"2026-08-27","scope":"Current public index; highest tested effort configuration"},{"benchmark":"SciCode","score":25.9,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-flash","measuredAt":"2026-08-27","scope":"Scientific coding evaluation"},{"benchmark":"AA Omniscience Index","score":-48.5,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-flash","measuredAt":"2026-08-27","scope":"Current public knowledge and hallucination index"},{"benchmark":"MLCR overall","score":18.3,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-flash","measuredAt":"2026-08-27","scope":"Machine-learning code reasoning evaluation"},{"benchmark":"GDPval-AA normalized","score":17.2,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-flash","measuredAt":"2026-08-27","scope":"Normalized real-world knowledge-work evaluation"},{"benchmark":"τ²-bench","score":83.9,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-flash","measuredAt":"2026-08-27","scope":"Tool-agent benchmark across supported domains"},{"benchmark":"Terminal-Bench Hard","score":25.8,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-flash","measuredAt":"2026-08-27","scope":"Hard terminal-agent tasks"},{"benchmark":"Terminal-Bench 2.1","score":61.8,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-flash","measuredAt":"2026-08-27","scope":"Terminal-agent task completion"},{"benchmark":"LCR","score":35,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-flash","measuredAt":"2026-08-27","scope":"Long-context reasoning evaluation"},{"benchmark":"IFBench","score":39.9,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-flash","measuredAt":"2026-08-27","scope":"Instruction-following evaluation"},{"benchmark":"Humanity's Last Exam","score":8.6,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-flash","measuredAt":"2026-08-27","scope":"Broad expert-level knowledge evaluation"},{"benchmark":"GPQA","score":65.6,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-flash","measuredAt":"2026-08-27","scope":"Graduate-level science questions"},{"benchmark":"CritPt","score":0,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-flash","measuredAt":"2026-08-27","scope":"Critical-point reasoning evaluation"},{"benchmark":"LiveCodeBench","score":40.2,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-flash","measuredAt":"2026-08-27","scope":"Contamination-resistant coding tasks"},{"benchmark":"AIME 2025","score":67.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-flash","measuredAt":"2026-08-27","scope":"Competition mathematics evaluation"},{"benchmark":"Enterprise Ops Gym","score":50.1,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-flash","measuredAt":"2026-08-27","scope":"Enterprise operations agent tasks"},{"benchmark":"LMArena overall","score":1395,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 144 · 10,982 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"xiaomi/mimo-v2.5-20260422","name":"MiMo-V2.5","releaseLabel":"V2.5","generationId":"v2.5","generationLabel":"V2.5","variantLabel":"Base","createdAt":"2026-04-22T16:11:09.000Z","kind":"multimodal","contextLength":1050000,"inputModalities":["text","audio","image","video"],"outputModalities":["text"],"promptPerMillion":0.14,"completionPerMillion":0.28,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...","huggingFaceId":"XiaomiMiMo/MiMo-V2.5","routes":[{"provider":"OpenRouter","modelId":"xiaomi/mimo-v2.5","url":"https://openrouter.ai/xiaomi/mimo-v2.5","free":false,"batch":false}],"benchmarks":[{"benchmark":"AA Intelligence Index","score":38,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-0424","measuredAt":"2026-08-27","scope":"Current public index; highest tested effort configuration"},{"benchmark":"AA Agentic Index","score":24.4,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-0424","measuredAt":"2026-08-27","scope":"Current public agentic composite"},{"benchmark":"SciCode","score":43.1,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-0424","measuredAt":"2026-08-27","scope":"Scientific coding evaluation"},{"benchmark":"AA Omniscience Index","score":-9.8,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-0424","measuredAt":"2026-08-27","scope":"Current public knowledge and hallucination index"},{"benchmark":"GDPval-AA normalized","score":32.4,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-0424","measuredAt":"2026-08-27","scope":"Normalized real-world knowledge-work evaluation"},{"benchmark":"τ²-bench","score":90.6,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-0424","measuredAt":"2026-08-27","scope":"Tool-agent benchmark across supported domains"},{"benchmark":"τ²-bench Banking","score":8.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-0424","measuredAt":"2026-08-27","scope":"Tool-agent benchmark in the banking domain"},{"benchmark":"Terminal-Bench Hard","score":41.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-0424","measuredAt":"2026-08-27","scope":"Hard terminal-agent tasks"},{"benchmark":"Terminal-Bench 2.1","score":63.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-0424","measuredAt":"2026-08-27","scope":"Terminal-agent task completion"},{"benchmark":"LCR","score":68.3,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-0424","measuredAt":"2026-08-27","scope":"Long-context reasoning evaluation"},{"benchmark":"IFBench","score":67.1,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-0424","measuredAt":"2026-08-27","scope":"Instruction-following evaluation"},{"benchmark":"Humanity's Last Exam","score":27.2,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-0424","measuredAt":"2026-08-27","scope":"Broad expert-level knowledge evaluation"},{"benchmark":"GPQA","score":84.9,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-0424","measuredAt":"2026-08-27","scope":"Graduate-level science questions"},{"benchmark":"CritPt","score":3.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-0424","measuredAt":"2026-08-27","scope":"Critical-point reasoning evaluation"},{"benchmark":"MMMU Pro","score":75.4,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-0424","measuredAt":"2026-08-27","scope":"Multimodal expert reasoning evaluation"},{"benchmark":"LMArena overall","score":1427,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 86 · 44,557 votes"},{"benchmark":"ResearchClawBench · overall","score":16.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5","measuredAt":"2026-05-15","scope":"Registered model evaluation · Model Card"},{"benchmark":"Claw-Eval · general","score":62.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://claw-eval.github.io","measuredAt":"2026-04-23","scope":"Registered model evaluation · Claw-Eval Leaderboard"},{"benchmark":"Claw-Eval · multimodal","score":23.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://claw-eval.github.io","measuredAt":"2026-04-23","scope":"Registered model evaluation · Claw-Eval Leaderboard"},{"benchmark":"Claw-Eval · multi turn","score":63.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://claw-eval.github.io","measuredAt":"2026-04-23","scope":"Registered model evaluation · Claw-Eval Leaderboard"},{"benchmark":"SWE-bench_Pro · SWE Bench Pro","score":56.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5","measuredAt":"2026-04-27","scope":"Registered model evaluation · Model Card"},{"benchmark":"terminal-bench-2.0 · terminalbench 2","score":65.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5","measuredAt":"2026-04-27","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"xiaomi/mimo-v2.5-pro-20260422","name":"MiMo-V2.5-Pro","releaseLabel":"V2.5-Pro","generationId":"v2.5","generationLabel":"V2.5","variantLabel":"Pro","createdAt":"2026-04-22T16:11:13.000Z","kind":"language","contextLength":1050000,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.435,"completionPerMillion":0.87,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....","huggingFaceId":"XiaomiMiMo/MiMo-V2.5-Pro","routes":[{"provider":"OpenRouter","modelId":"xiaomi/mimo-v2.5-pro","url":"https://openrouter.ai/xiaomi/mimo-v2.5-pro","free":false,"batch":false}],"benchmarks":[{"benchmark":"AA Intelligence Index","score":42.9,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-pro","measuredAt":"2026-08-27","scope":"Current public index; highest tested effort configuration"},{"benchmark":"AA Agentic Index","score":29.5,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-pro","measuredAt":"2026-08-27","scope":"Current public agentic composite"},{"benchmark":"SciCode","score":50.2,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-pro","measuredAt":"2026-08-27","scope":"Scientific coding evaluation"},{"benchmark":"AA Omniscience Index","score":3.3,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-pro","measuredAt":"2026-08-27","scope":"Current public knowledge and hallucination index"},{"benchmark":"MLCR overall","score":9.4,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-pro","measuredAt":"2026-08-27","scope":"Machine-learning code reasoning evaluation"},{"benchmark":"GDPval-AA normalized","score":38.1,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-pro","measuredAt":"2026-08-27","scope":"Normalized real-world knowledge-work evaluation"},{"benchmark":"ITBench SRE","score":38.2,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-pro","measuredAt":"2026-08-27","scope":"Site-reliability engineering agent tasks"},{"benchmark":"τ²-bench","score":94.2,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-pro","measuredAt":"2026-08-27","scope":"Tool-agent benchmark across supported domains"},{"benchmark":"τ²-bench Banking","score":9.9,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-pro","measuredAt":"2026-08-27","scope":"Tool-agent benchmark in the banking domain"},{"benchmark":"Terminal-Bench Hard","score":43.2,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-pro","measuredAt":"2026-08-27","scope":"Hard terminal-agent tasks"},{"benchmark":"Terminal-Bench 2.1","score":65.2,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-pro","measuredAt":"2026-08-27","scope":"Terminal-agent task completion"},{"benchmark":"LCR","score":77.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-pro","measuredAt":"2026-08-27","scope":"Long-context reasoning evaluation"},{"benchmark":"IFBench","score":79.9,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-pro","measuredAt":"2026-08-27","scope":"Instruction-following evaluation"},{"benchmark":"Humanity's Last Exam","score":35.7,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-pro","measuredAt":"2026-08-27","scope":"Broad expert-level knowledge evaluation"},{"benchmark":"GPQA","score":86.6,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-pro","measuredAt":"2026-08-27","scope":"Graduate-level science questions"},{"benchmark":"CritPt","score":4,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-pro","measuredAt":"2026-08-27","scope":"Critical-point reasoning evaluation"},{"benchmark":"APEX Agents","score":2.4,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-pro","measuredAt":"2026-08-27","scope":"Professional agent task evaluation"},{"benchmark":"Analyst Agent","score":20,"unit":"%","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mimo-v2-5-pro","measuredAt":"2026-08-27","scope":"Analyst workflow agent tasks"},{"benchmark":"LMArena overall","score":1465,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 28 · 54,177 votes"},{"benchmark":"WildClawBench · overall","score":43,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://internlm.github.io/WildClawBench","measuredAt":"2026-05-22","scope":"Registered model evaluation · WildClawBench"},{"benchmark":"Claw-Eval · general","score":64,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://claw-eval.github.io","measuredAt":"2026-04-23","scope":"Registered model evaluation · Claw-Eval Leaderboard"},{"benchmark":"Claw-Eval · multi turn","score":63.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://claw-eval.github.io","measuredAt":"2026-04-23","scope":"Registered model evaluation · Claw-Eval Leaderboard"},{"benchmark":"gpqa · diamond","score":66.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5-Pro","measuredAt":"2026-04-27","scope":"Registered model evaluation · Model Card"},{"benchmark":"gsm8k · gsm8k","score":99.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5-Pro","measuredAt":"2026-04-27","scope":"Registered model evaluation · Model Card"},{"benchmark":"hle · hle","score":48,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5-Pro","measuredAt":"2026-04-27","scope":"Registered model evaluation · Model Card"},{"benchmark":"MMLU-Pro · mmlu pro","score":68.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5-Pro","measuredAt":"2026-04-27","scope":"Registered model evaluation · Model Card"},{"benchmark":"SWE-bench_Pro · SWE Bench Pro","score":57.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5-Pro","measuredAt":"2026-04-27","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]}]},{"id":"arcee-ai","name":"Arcee AI","domain":"arcee.ai","major":false,"modelCount":2,"series":[{"id":"trinity","label":"Trinity","models":[{"id":"arcee-ai/trinity-large-thinking","name":"Trinity Large Thinking","releaseLabel":"Large Thinking","generationId":"releases","generationLabel":"Releases","variantLabel":"Large Thinking","createdAt":"2026-04-01T15:45:18.000Z","kind":"language","contextLength":262144,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.22,"completionPerMillion":0.85,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Trinity Large Thinking is a powerful open source reasoning model from the team at Arcee AI. It shows strong performance in PinchBench, agentic workloads, and reasoning tasks. Launch video: https://youtu.be/Gc82AXLa0Rg?si=4RLn6WBz33qT--B7...","huggingFaceId":"arcee-ai/Trinity-Large-Thinking","routes":[{"provider":"OpenRouter","modelId":"arcee-ai/trinity-large-thinking","url":"https://openrouter.ai/arcee-ai/trinity-large-thinking","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1342,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 192 · 28,994 votes"},{"benchmark":"gpqa · diamond","score":76.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/arcee-ai/Trinity-Large-Thinking","measuredAt":"2026-04-02","scope":"Registered model evaluation · Model Card"},{"benchmark":"MMLU-Pro · mmlu pro","score":83.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/arcee-ai/Trinity-Large-Thinking","measuredAt":"2026-04-02","scope":"Registered model evaluation · Model Card"},{"benchmark":"SWE-bench_Verified · swe bench % resolved","score":63.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/arcee-ai/Trinity-Large-Thinking","measuredAt":"2026-04-02","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]},{"id":"virtuoso","label":"Virtuoso","models":[{"id":"arcee-ai/virtuoso-large","name":"Virtuoso Large","releaseLabel":"Large","generationId":"releases","generationLabel":"Releases","variantLabel":"Large","createdAt":"2025-05-05T21:01:25.000Z","kind":"language","contextLength":131072,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.75,"completionPerMillion":1.2,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Virtuoso‑Large is Arcee's top‑tier general‑purpose LLM at 72 B parameters, tuned to tackle cross‑domain reasoning, creative writing and enterprise QA. Unlike many 70 B peers, it retains the 128 k...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"arcee-ai/virtuoso-large","url":"https://openrouter.ai/arcee-ai/virtuoso-large","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]}]},{"id":"ibm-granite","name":"IBM","domain":"ibm.com","major":false,"modelCount":2,"series":[{"id":"granite","label":"Granite","models":[{"id":"ibm-granite/granite-4.0-h-micro","name":"Granite 4.0 Micro","releaseLabel":"4.0 Micro","generationId":"4.0","generationLabel":"4.0","variantLabel":"H Micro","createdAt":"2025-10-20T02:34:55.000Z","kind":"language","contextLength":131000,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.017,"completionPerMillion":0.112,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Granite-4.0-H-Micro is a 3B parameter from the Granite 4 family of models. These models are the latest in a series of models released by IBM. They are fine-tuned for long...","huggingFaceId":"ibm-granite/granite-4.0-h-micro","routes":[{"provider":"OpenRouter","modelId":"ibm-granite/granite-4.0-h-micro","url":"https://openrouter.ai/ibm-granite/granite-4.0-h-micro","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"ibm-granite/granite-4.1-8b-20260429","name":"Granite 4.1 8B","releaseLabel":"4.1 8B","generationId":"4.1","generationLabel":"4.1","variantLabel":"8B","createdAt":"2026-04-30T19:24:31.000Z","kind":"language","contextLength":131072,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.049999999999999996,"completionPerMillion":0.09999999999999999,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Granite 4.1 8B is a dense, decoder-only 8-billion-parameter language model from IBM, part of the Granite 4.1 family. It supports a 131K-token context window and is designed for enterprise tasks...","huggingFaceId":"ibm-granite/granite-4.1-8b","routes":[{"provider":"OpenRouter","modelId":"ibm-granite/granite-4.1-8b","url":"https://openrouter.ai/ibm-granite/granite-4.1-8b","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1292,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 234 · 4,034 votes"},{"benchmark":"ifstruct-v1.0 · ifstruct v1","score":68.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://www.liquid.ai/blog/ifstruct-v1.0","measuredAt":"2026-06-30","scope":"Registered model evaluation · Liquid AI — IFStruct v1.0 blog (granite-4.1-8b)"},{"benchmark":"gsm8k · gsm8k","score":92.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/ibm-granite/granite-4.1-8b","measuredAt":"2026-04-29","scope":"Registered model evaluation · Model Card"},{"benchmark":"MMLU-Pro · mmlu pro","score":56,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/ibm-granite/granite-4.1-8b","measuredAt":"2026-04-29","scope":"Registered model evaluation · Model Card"},{"benchmark":"gpqa · diamond","score":42,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/ibm-granite/granite-4.1-8b","measuredAt":"2026-04-29","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]}]},{"id":"inclusionai","name":"Inclusionai","domain":null,"major":false,"modelCount":2,"series":[{"id":"ling","label":"Ling","models":[{"id":"inclusionai/ling-3.0-flash-20260723","name":"Ling-3.0-flash","releaseLabel":"3.0-flash","generationId":"3.0","generationLabel":"3.0","variantLabel":"Flash","createdAt":"2026-07-23T14:56:20.000Z","kind":"language","contextLength":262144,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.020999999999999998,"completionPerMillion":0.063,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"*Ling-3.0-flash* is a *124B-parameter Mixture-of-Experts (MoE) model*, with approximately *5.1B parameters activated per token*. The model is designed with *token efficiency and production-scale agentic inference* as key priorities, enabling developers...","huggingFaceId":"inclusionAI/Ling-3.0-flash","routes":[{"provider":"OpenRouter","modelId":"inclusionai/ling-3.0-flash","url":"https://openrouter.ai/inclusionai/ling-3.0-flash","free":false,"batch":false}],"benchmarks":[{"benchmark":"aime_2026 · MathArena/aime 2026","score":93.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/inclusionAI/Ling-3.0-flash","measuredAt":"2026-08-06","scope":"Registered model evaluation · Model Card"},{"benchmark":"hle · hle","score":22.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/inclusionAI/Ling-3.0-flash","measuredAt":"2026-08-06","scope":"Registered model evaluation · Model Card"},{"benchmark":"hmmt_feb_2026 · MathArena/hmmt feb 2026","score":87,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/inclusionAI/Ling-3.0-flash","measuredAt":"2026-08-06","scope":"Registered model evaluation · Model Card"},{"benchmark":"SWE-bench_Multilingual · swe bench multilingual % resolved","score":72.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/inclusionAI/Ling-3.0-flash","measuredAt":"2026-08-06","scope":"Registered model evaluation · Model Card"},{"benchmark":"SWE-bench_Pro · SWE Bench Pro","score":56.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/inclusionAI/Ling-3.0-flash","measuredAt":"2026-08-06","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"eedf22ce-d818-4f73-8f87-643a17565b9a","title":"Services: The New Software","href":"/intel/eedf22ce-d818-4f73-8f87-643a17565b9a","hook":"Sequoia's argument that AI service companies selling outcomes will capture more value than tools selling software. Read it for the strategic frame if you are…"},{"id":"39eb0f39-568d-46dd-8ef8-3b716bd121d2","title":"Deep Agents Evaluation Framework · article","href":"/intel/39eb0f39-568d-46dd-8ef8-3b716bd121d2","hook":"A systematic framework for curating eval data and measuring agent behavior, instead of eyeballing whether it \"seems better.\" Read it if you want your agent i…"},{"id":"1d6f94f8-1997-4c8c-9612-7c573243bcb4","title":"Imagine if naked people were stupider. It turns out, naked models actually are.","href":"/intel/1d6f94f8-1997-4c8c-9612-7c573243bcb4","hook":"Jepsen's Kyle Kingsbury turns his distributed-systems rigor on LLMs, and the results are humbling. Read it for a clear-eyed, adversarial view of where models…"}]},{"id":"inclusionai/ling-3.0-flash-fin-20260827","name":"Ling 3.0 Flash Fin","releaseLabel":"3.0 Flash Fin","generationId":"3.0","generationLabel":"3.0","variantLabel":"Flash Fin","createdAt":"2026-08-27T15:58:10.000Z","kind":"language","contextLength":262144,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Ling 3.0 Flash Fin is a finance-focused mixture-of-experts model from InclusionAI, built on Ling 3.0 Flash with 5.1B active parameters out of 124B total. It is designed for real-world investment...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"inclusionai/ling-3.0-flash-fin:free","url":"https://openrouter.ai/inclusionai/ling-3.0-flash-fin:free","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"eedf22ce-d818-4f73-8f87-643a17565b9a","title":"Services: The New Software","href":"/intel/eedf22ce-d818-4f73-8f87-643a17565b9a","hook":"Sequoia's argument that AI service companies selling outcomes will capture more value than tools selling software. Read it for the strategic frame if you are…"},{"id":"39eb0f39-568d-46dd-8ef8-3b716bd121d2","title":"Deep Agents Evaluation Framework · article","href":"/intel/39eb0f39-568d-46dd-8ef8-3b716bd121d2","hook":"A systematic framework for curating eval data and measuring agent behavior, instead of eyeballing whether it \"seems better.\" Read it if you want your agent i…"},{"id":"1d6f94f8-1997-4c8c-9612-7c573243bcb4","title":"Imagine if naked people were stupider. It turns out, naked models actually are.","href":"/intel/1d6f94f8-1997-4c8c-9612-7c573243bcb4","hook":"Jepsen's Kyle Kingsbury turns his distributed-systems rigor on LLMs, and the results are humbling. Read it for a clear-eyed, adversarial view of where models…"}]}]}]},{"id":"liquid","name":"Liquid AI","domain":"liquid.ai","major":false,"modelCount":2,"series":[{"id":"lfm","label":"LFM","models":[{"id":"liquid/lfm-2.5-2.6b-20260811","name":"LFM2.5-2.6B","releaseLabel":"2.5-2.6B","generationId":"2.5","generationLabel":"2.5","variantLabel":"2.6B","createdAt":"2026-08-11T17:48:39.000Z","kind":"language","contextLength":65536,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"LFM2.5-2.6B is a compact reasoning model from Liquid AI. It is suited for agent workflows, data extraction, RAG, and long-context processing. Liquid advises against using it for agentic coding or...","huggingFaceId":"LiquidAI/LFM2.5-2.6B","routes":[{"provider":"OpenRouter","modelId":"liquid/lfm-2.5-2.6b:free","url":"https://openrouter.ai/liquid/lfm-2.5-2.6b:free","free":true,"batch":false}],"benchmarks":[{"benchmark":"ifstruct-v1.0 · ifstruct v1","score":85.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://www.liquid.ai/blog/lfm2-5-2-6b","measuredAt":"2026-07-28","scope":"Registered model evaluation · Liquid AI — LFM2.5-2.6B blog"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"liquid/lfm-2.5-embedding-350m-20260818","name":"LFM2.5-Embedding-350M","releaseLabel":"2.5-Embedding-350M","generationId":"2.5","generationLabel":"2.5","variantLabel":"Embedding 350m","createdAt":"2026-08-18T18:31:48.000Z","kind":"embedding","contextLength":512,"inputModalities":["text"],"outputModalities":["embeddings"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"LFM2.5-Embedding-350M is a text embedding model from Liquid AI. It produces 1,024-dimensional embeddings for retrieval and semantic search. Successful OpenRouter requests and embeddings may be retained and used to train...","huggingFaceId":"LiquidAI/LFM2.5-Embedding-350M","routes":[{"provider":"OpenRouter","modelId":"liquid/lfm-2.5-embedding-350m:free","url":"https://openrouter.ai/liquid/lfm-2.5-embedding-350m:free","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[]}]}]},{"id":"morph","name":"Morph","domain":"morphllm.com","major":false,"modelCount":2,"series":[{"id":"morph","label":"Morph","models":[{"id":"morph/morph-v3-fast","name":"Morph V3 Fast","releaseLabel":"V3 Fast","generationId":"v3","generationLabel":"V3","variantLabel":"Fast","createdAt":"2025-07-07T17:40:02.000Z","kind":"language","contextLength":81920,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.7999999999999999,"completionPerMillion":1.2,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Morph's fastest apply model for code edits. ~10,500 tokens/sec with 96% accuracy for rapid code transformations. The model requires the prompt to be in the following format: <instruction>{instruction}</instruction> <code>{initial_code}</code> <update>{edit_snippet}</update>...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"morph/morph-v3-fast","url":"https://openrouter.ai/morph/morph-v3-fast","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"morph/morph-v3-large","name":"Morph V3 Large","releaseLabel":"V3 Large","generationId":"v3","generationLabel":"V3","variantLabel":"Large","createdAt":"2025-07-07T17:54:18.000Z","kind":"language","contextLength":262144,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.8999999999999999,"completionPerMillion":1.9,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Morph's high-accuracy apply model for complex code edits. ~4,500 tokens/sec with 98% accuracy for precise code transformations. The model requires the prompt to be in the following format: <instruction>{instruction}</instruction> <code>{initial_code}</code>...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"morph/morph-v3-large","url":"https://openrouter.ai/morph/morph-v3-large","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]}]},{"id":"nex-agi","name":"NEX AGI","domain":"nex-agi.com","major":false,"modelCount":2,"series":[{"id":"nex","label":"NEX","models":[{"id":"nex-agi/nex-n2-pro","name":"Nex-N2-Pro","releaseLabel":"N2-Pro","generationId":"releases","generationLabel":"Releases","variantLabel":"N2 Pro","createdAt":"2026-06-08T16:45:40.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0.25,"completionPerMillion":1,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Nex-N2-Pro is an agentic mixture-of-experts model from Nex AGI, with 17B active parameters out of 397B total. Built on the Qwen3.5 architecture, it accepts text and image input and produces...","huggingFaceId":"nex-agi/Nex-N2-Pro","routes":[{"provider":"OpenRouter","modelId":"nex-agi/nex-n2-pro","url":"https://openrouter.ai/nex-agi/nex-n2-pro","free":false,"batch":false}],"benchmarks":[{"benchmark":"WildClawBench · overall","score":53.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://internlm.github.io/WildClawBench","measuredAt":"2026-05-22","scope":"Registered model evaluation · WildClawBench"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"6a16b2bd-50a1-4469-b9f7-80e552960eaa","title":"Harness Engineering Guide","href":"/intel/6a16b2bd-50a1-4469-b9f7-80e552960eaa","hook":"An open, growing reference for building agent runtimes — concepts, tutorials, papers, tools — in one curated place. The map to start from if harness engineer…"},{"id":"c1f469e7-8a5e-4722-bf30-d8dc83c310db","title":"Rhys's MCP Retrospective","href":"/intel/c1f469e7-8a5e-4722-bf30-d8dc83c310db","hook":"Rhys Sullivan's analysis of why MCP didn't take off when it launched alongside models like Sonnet 3.5 and GPT-4o, and what comes next for agent-to-tool conne…"},{"id":"e9fdb80b-b463-4766-99ed-c84d437714d0","title":"Anthropic's Boris Cherny: Why Coding Is Solved, and What Comes Next","href":"/intel/e9fdb80b-b463-4766-99ed-c84d437714d0","hook":"Cherny's claim isn't that AI helps you code faster—it's that the bottleneck has already moved off your keyboard entirely, toward dispatching and reviewing ag…"}]},{"id":"nex-agi/nex-n2-mini","name":"Nex-N2-Mini","releaseLabel":"N2-Mini","generationId":"releases","generationLabel":"Releases","variantLabel":"N2 Mini","createdAt":"2026-06-24T14:56:04.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0.024999999999999998,"completionPerMillion":0.09999999999999999,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Nex-N2-Mini is an open-source agentic mixture-of-experts model from Nex AGI, the smaller sibling in the Nex-N2 series. It accepts text and image input and is built for coding, tool use,...","huggingFaceId":"nex-agi/Nex-N2-Mini","routes":[{"provider":"OpenRouter","modelId":"nex-agi/nex-n2-mini","url":"https://openrouter.ai/nex-agi/nex-n2-mini","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"6a16b2bd-50a1-4469-b9f7-80e552960eaa","title":"Harness Engineering Guide","href":"/intel/6a16b2bd-50a1-4469-b9f7-80e552960eaa","hook":"An open, growing reference for building agent runtimes — concepts, tutorials, papers, tools — in one curated place. The map to start from if harness engineer…"},{"id":"c1f469e7-8a5e-4722-bf30-d8dc83c310db","title":"Rhys's MCP Retrospective","href":"/intel/c1f469e7-8a5e-4722-bf30-d8dc83c310db","hook":"Rhys Sullivan's analysis of why MCP didn't take off when it launched alongside models like Sonnet 3.5 and GPT-4o, and what comes next for agent-to-tool conne…"},{"id":"e9fdb80b-b463-4766-99ed-c84d437714d0","title":"Anthropic's Boris Cherny: Why Coding Is Solved, and What Comes Next","href":"/intel/e9fdb80b-b463-4766-99ed-c84d437714d0","hook":"Cherny's claim isn't that AI helps you code faster—it's that the bottleneck has already moved off your keyboard entirely, toward dispatching and reviewing ag…"}]}]}]},{"id":"poolside","name":"Poolside","domain":"poolside.ai","major":false,"modelCount":2,"series":[{"id":"laguna","label":"Laguna","models":[{"id":"poolside/laguna-xs-2.1-20260625","name":"Laguna XS 2.1","releaseLabel":"XS 2.1","generationId":"2.1","generationLabel":"2.1","variantLabel":"Xs","createdAt":"2026-07-02T14:27:09.000Z","kind":"language","contextLength":262144,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.06,"completionPerMillion":0.12,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Laguna XS 2.1 is the latest coding agent model in the 33B-A3B category from [Poolside](https://poolside.ai/) and a step forward from their Laguna XS.2 model (released in April 2026). It combines...","huggingFaceId":"poolside/Laguna-XS-2.1","routes":[{"provider":"OpenRouter","modelId":"poolside/laguna-xs-2.1","url":"https://openrouter.ai/poolside/laguna-xs-2.1","free":false,"batch":false},{"provider":"OpenRouter","modelId":"poolside/laguna-xs-2.1:free","url":"https://openrouter.ai/poolside/laguna-xs-2.1:free","free":true,"batch":false}],"benchmarks":[{"benchmark":"SWE-bench_Pro · SWE Bench Pro","score":47.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/poolside/Laguna-XS-2.1","measuredAt":"2026-07-02","scope":"Registered model evaluation · Model Card"},{"benchmark":"SWE-bench_Verified · swe bench % resolved","score":70.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/poolside/Laguna-XS-2.1","measuredAt":"2026-07-02","scope":"Registered model evaluation · Model Card"},{"benchmark":"terminal-bench-2.0 · terminalbench 2","score":37.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/poolside/Laguna-XS-2.1","measuredAt":"2026-07-02","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"bc3e57e7-be1a-4585-a43a-df62724ccd70","title":"Latest open artifacts (#22): Zyphra, Cohere, and Poolside are expanding the breadth of the ecosystem","href":"/intel/bc3e57e7-be1a-4585-a43a-df62724ccd70","hook":"Most open-model coverage fixates on the frontier race, but this tracks a quieter structural shift: the release landscape has broadened from a handful of most…"}]},{"id":"poolside/laguna-s-2.1-20260720","name":"Laguna S 2.1","releaseLabel":"S 2.1","generationId":"2.1","generationLabel":"2.1","variantLabel":"S","createdAt":"2026-07-21T16:51:23.000Z","kind":"language","contextLength":1048576,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.09,"completionPerMillion":0.18,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Laguna S 2.1 is the latest coding agent model from [Poolside](<https://poolside.ai/>). Laguna S 2.1 is a 118B total parameter model with 8B active parameters, scoring 70.2% on Terminal-Bench 2.1 and...","huggingFaceId":"poolside/Laguna-S-2.1","routes":[{"provider":"OpenRouter","modelId":"poolside/laguna-s-2.1","url":"https://openrouter.ai/poolside/laguna-s-2.1","free":false,"batch":false},{"provider":"OpenRouter","modelId":"poolside/laguna-s-2.1:free","url":"https://openrouter.ai/poolside/laguna-s-2.1:free","free":true,"batch":false}],"benchmarks":[{"benchmark":"deep-swe · deep swe","score":40.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/poolside/Laguna-S-2.1","measuredAt":"2026-07-21","scope":"Registered model evaluation · Model Card"},{"benchmark":"SWE-bench_Multilingual · swe bench % resolved","score":78.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/poolside/Laguna-S-2.1","measuredAt":"2026-07-21","scope":"Registered model evaluation · Model Card"},{"benchmark":"SWE-bench_Pro · SWE Bench Pro","score":59.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/poolside/Laguna-S-2.1","measuredAt":"2026-07-21","scope":"Registered model evaluation · Model Card"},{"benchmark":"terminal-bench-2.1 · terminalbench 2 1","score":70.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/poolside/Laguna-S-2.1","measuredAt":"2026-08-26","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"bc3e57e7-be1a-4585-a43a-df62724ccd70","title":"Latest open artifacts (#22): Zyphra, Cohere, and Poolside are expanding the breadth of the ecosystem","href":"/intel/bc3e57e7-be1a-4585-a43a-df62724ccd70","hook":"Most open-model coverage fixates on the frontier race, but this tracks a quieter structural shift: the release landscape has broadened from a handful of most…"}]}]}]},{"id":"rekaai","name":"Reka AI","domain":"reka.ai","major":false,"modelCount":2,"series":[{"id":"reka","label":"Reka","models":[{"id":"rekaai/reka-flash-3","name":"Reka Flash 3","releaseLabel":"Flash 3","generationId":"3","generationLabel":"3","variantLabel":"Flash","createdAt":"2025-03-12T20:53:33.000Z","kind":"language","contextLength":65536,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.09999999999999999,"completionPerMillion":0.19999999999999998,"imageOutput":null,"supportsTools":false,"supportsReasoning":true,"description":"Reka Flash 3 is a general-purpose, instruction-tuned large language model with 21 billion parameters, developed by Reka. It excels at general chat, coding tasks, instruction-following, and function calling. Featuring a...","huggingFaceId":"RekaAI/reka-flash-3","routes":[{"provider":"OpenRouter","modelId":"rekaai/reka-flash-3","url":"https://openrouter.ai/rekaai/reka-flash-3","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"rekaai/reka-edge-2603","name":"Reka Edge","releaseLabel":"Edge","generationId":"releases","generationLabel":"Releases","variantLabel":"Edge","createdAt":"2026-03-20T17:16:05.000Z","kind":"multimodal","contextLength":16384,"inputModalities":["image","text","video"],"outputModalities":["text"],"promptPerMillion":0.09999999999999999,"completionPerMillion":0.09999999999999999,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"Reka Edge is an extremely efficient 7B multimodal vision-language model that accepts image/video+text inputs and generates text outputs. This model is optimized specifically to deliver industry-leading performance in image understanding,...","huggingFaceId":"RekaAI/reka-edge-2603","routes":[{"provider":"OpenRouter","modelId":"rekaai/reka-edge","url":"https://openrouter.ai/rekaai/reka-edge","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]}]},{"id":"relace","name":"Relace","domain":"relace.ai","major":false,"modelCount":2,"series":[{"id":"relace","label":"Relace","models":[{"id":"relace/relace-apply-3","name":"Relace Apply 3","releaseLabel":"Apply 3","generationId":"3","generationLabel":"3","variantLabel":"Apply","createdAt":"2025-09-26T12:59:32.000Z","kind":"language","contextLength":256000,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.85,"completionPerMillion":1.25,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Relace Apply 3 is a specialized code-patching LLM that merges AI-suggested edits straight into your source files. It can apply updates from GPT-4o, Claude, and others into your files at...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"relace/relace-apply-3","url":"https://openrouter.ai/relace/relace-apply-3","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"relace/relace-search-20251208","name":"Relace Search","releaseLabel":"Search","generationId":"releases","generationLabel":"Releases","variantLabel":"Search","createdAt":"2025-12-08T17:06:00.000Z","kind":"language","contextLength":256000,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":1,"completionPerMillion":3,"imageOutput":null,"supportsTools":true,"supportsReasoning":false,"description":"The relace-search model uses 4-12 `view_file` and `grep` tools in parallel to explore a codebase and return relevant files to the user request. In contrast to RAG, relace-search performs agentic...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"relace/relace-search","url":"https://openrouter.ai/relace/relace-search","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]}]},{"id":"runway","name":"Runway","domain":"runwayml.com","major":false,"modelCount":2,"series":[{"id":"aleph","label":"Aleph","models":[{"id":"runway/aleph-2-20260729","name":"Aleph 2.0","releaseLabel":"2.0","generationId":"2","generationLabel":"2","variantLabel":".0","createdAt":"2026-07-29T15:38:04.000Z","kind":"video","contextLength":0,"inputModalities":["text","image","video"],"outputModalities":["video"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Runway Aleph 2.0 is an in-context video editing model from Runway. It applies text instructions and keyframe-guided edits across existing footage while preserving details that are not meant to change....","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"runway/aleph-2","url":"https://openrouter.ai/runway/aleph-2","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Video Arena","url":"https://artificialanalysis.ai/video/leaderboard/text-to-video","note":"Blind preference ranking for video generation"}],"intel":[]}]},{"id":"gen","label":"GEN","models":[{"id":"runway/gen-4.5-20260729","name":"Gen-4.5","releaseLabel":"4.5","generationId":"4.5","generationLabel":"4.5","variantLabel":"Base","createdAt":"2026-07-29T15:38:03.000Z","kind":"video","contextLength":0,"inputModalities":["text","image"],"outputModalities":["video"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Runway Gen-4.5 is a video generation model from Runway for text-to-video and image-to-video workflows. It is designed for cinematic scene creation with strong motion quality, visual fidelity, and prompt adherence....","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"runway/gen-4.5","url":"https://openrouter.ai/runway/gen-4.5","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Video Arena","url":"https://artificialanalysis.ai/video/leaderboard/text-to-video","note":"Blind preference ranking for video generation"}],"intel":[{"id":"159da36e-2834-4aac-b9d5-3d94aa123274","title":"TradingAgents","href":"/intel/159da36e-2834-4aac-b9d5-3d94aa123274","hook":"Most \"LLM trading bot\" projects are one prompt wrapped around a price feed. TradingAgents instead models an actual firm — separate agents for fundamentals, s…"},{"id":"f7f80504-c539-4937-a09e-7ff21d0a67fc","title":"ChatDev 2.0","href":"/intel/f7f80504-c539-4937-a09e-7ff21d0a67fc","hook":"Zero-code multi-agent orchestration: you describe the system in config and it builds and runs the agent team. Worth a look if you are tired of hand-wiring ag…"},{"id":"e4da4983-d8f3-41b2-b60b-48fab0eadc75","title":"Agent Memory Architecture Guide","href":"/intel/e4da4983-d8f3-41b2-b60b-48fab0eadc75","hook":"A technical progression from a Python list to graph-vector hybrid memory, with the tradeoffs at each step. The reference to reach for when \"just stuff it in …"}]}]}]},{"id":"sakana","name":"Sakana AI","domain":"sakana.ai","major":false,"modelCount":2,"series":[{"id":"fugu","label":"Fugu","models":[{"id":"sakana/fugu-ultra-20260615","name":"Fugu Ultra","releaseLabel":"Ultra","generationId":"releases","generationLabel":"Releases","variantLabel":"Ultra","createdAt":"2026-06-24T04:45:03.000Z","kind":"multimodal","contextLength":1000000,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":5,"completionPerMillion":30,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Fugu Ultra is the higher-performance model in Sakana AI's Fugu family. Rather than a single monolithic model, Fugu is a learned multi-agent orchestration system: a language model trained to route...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"sakana/fugu-ultra","url":"https://openrouter.ai/sakana/fugu-ultra","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]},{"id":"sakana","label":"Sakana","models":[{"id":"sakana/namazu-20260811","name":"Sakana Namazu","releaseLabel":"Namazu","generationId":"releases","generationLabel":"Releases","variantLabel":"Namazu","createdAt":"2026-08-11T01:02:09.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["text","image","file"],"outputModalities":["text"],"promptPerMillion":0.95,"completionPerMillion":4,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Sakana Namazu is a Japanese-specialized reasoning model from Sakana AI, based on Kimi K2.6 with additional training for Japanese language and business contexts. It is suited for Japanese instruction following,...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"sakana/sakana-namazu","url":"https://openrouter.ai/sakana/sakana-namazu","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]}]},{"id":"stepfun","name":"StepFun","domain":"stepfun.com","major":false,"modelCount":2,"series":[{"id":"step","label":"Step","models":[{"id":"stepfun/step-3.5-flash","name":"Step 3.5 Flash","releaseLabel":"3.5 Flash","generationId":"3.5","generationLabel":"3.5","variantLabel":"Flash","createdAt":"2026-01-29T23:12:17.000Z","kind":"language","contextLength":262144,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.09999999999999999,"completionPerMillion":0.3,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Step 3.5 Flash is StepFun's most capable open-source foundation model. Built on a sparse Mixture of Experts (MoE) architecture, it selectively activates only 11B of its 196B parameters per token....","huggingFaceId":"stepfun-ai/Step-3.5-Flash","routes":[{"provider":"OpenRouter","modelId":"stepfun/step-3.5-flash","url":"https://openrouter.ai/stepfun/step-3.5-flash","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1404,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 134 · 57,417 votes"},{"benchmark":"AIME 2026","score":96.7,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"GPQA","score":83.5,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"Humanity's Last Exam","score":23.1,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"HMMT 2026","score":86.4,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"MMLU-Pro","score":84.4,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"SWE-bench Verified","score":74.4,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"Terminal-Bench","score":51,"unit":"%","sourceName":"Hugging Face OpenEvals","sourceUrl":"https://huggingface.co/datasets/OpenEvals/leaderboard-data","measuredAt":"2026-08-27","scope":"Latest published official benchmark result aggregated by OpenEvals"},{"benchmark":"gpqa · diamond","score":83.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://arxiv.org/abs/2602.10604","measuredAt":"2026-02-11","scope":"Registered model evaluation · Step 3.5 Flash Paper"},{"benchmark":"hle · hle","score":23.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://arxiv.org/abs/2602.10604","measuredAt":"2026-02-11","scope":"Registered model evaluation · Step 3.5 Flash Paper"},{"benchmark":"MMLU-Pro · mmlu pro","score":84.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://arxiv.org/abs/2602.10604","measuredAt":"2026-02-11","scope":"Registered model evaluation · Step 3.5 Flash Paper"},{"benchmark":"SWE-bench_Verified · swe bench % resolved","score":74.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://arxiv.org/abs/2602.10604","measuredAt":"2026-02-11","scope":"Registered model evaluation · Step 3.5 Flash Paper"},{"benchmark":"terminal-bench-2.0 · terminalbench 2","score":51,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://arxiv.org/abs/2602.10604","measuredAt":"2026-02-11","scope":"Registered model evaluation · Step 3.5 Flash Paper"},{"benchmark":"WildClawBench · overall","score":26.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://internlm.github.io/WildClawBench","measuredAt":"2026-05-22","scope":"Registered model evaluation · WildClawBench"},{"benchmark":"hmmt_feb_2026 · MathArena/hmmt feb 2026","score":86.4,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://matharena.ai/?comp=hmmt--hmmt_feb_2026","measuredAt":"2026-02-23","scope":"Registered model evaluation · Official MathArena Evaluation"},{"benchmark":"aime_2026 · MathArena/aime 2026","score":96.7,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://matharena.ai/?comp=aime--aime_2026","measuredAt":"2026-02-16","scope":"Registered model evaluation · Official MathArena Evaluation"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"e4da4983-d8f3-41b2-b60b-48fab0eadc75","title":"Agent Memory Architecture Guide","href":"/intel/e4da4983-d8f3-41b2-b60b-48fab0eadc75","hook":"A technical progression from a Python list to graph-vector hybrid memory, with the tradeoffs at each step. The reference to reach for when \"just stuff it in …"},{"id":"c7c0bf34-6301-4f65-9bad-84e1821a7eec","title":"Prompt caching in LLMs, clearly explained","href":"/intel/c7c0bf34-6301-4f65-9bad-84e1821a7eec","hook":"A concrete teardown of how Claude hits a 92% cache hit-rate, and why it matters: every agent step resends the whole history, so caching is the line between a…"},{"id":"0e02d304-1b0c-470b-bb55-1cf6b544d3d1","title":"GPT-5","href":"/intel/0e02d304-1b0c-470b-bb55-1cf6b544d3d1","hook":"GPT-5's default toward autonomous, multi-step action on vague prompts changes how you scope tasks for it — Mollick's hands-on examples show where that proact…"}]},{"id":"stepfun/step-3.7-flash-20260528","name":"Step 3.7 Flash","releaseLabel":"3.7 Flash","generationId":"3.7","generationLabel":"3.7","variantLabel":"Flash","createdAt":"2026-05-28T16:17:49.000Z","kind":"multimodal","contextLength":262144,"inputModalities":["text","image","video"],"outputModalities":["text"],"promptPerMillion":0.19999999999999998,"completionPerMillion":1.15,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Step 3.7 Flash is StepFun's latest high-efficiency multimodal Mixture-of-Experts model. It pairs a 196B-parameter language backbone with a vision encoder for native image and video understanding, activating roughly 11B parameters...","huggingFaceId":"stepfun-ai/Step-3.7-Flash","routes":[{"provider":"OpenRouter","modelId":"stepfun/step-3.7-flash","url":"https://openrouter.ai/stepfun/step-3.7-flash","free":false,"batch":false}],"benchmarks":[{"benchmark":"terminal-bench-2.1 · terminalbench 2 1","score":59.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/stepfun-ai/Step-3.7-Flash","measuredAt":"2026-06-03","scope":"Registered model evaluation · Model Card"},{"benchmark":"SWE-bench_Pro · SWE Bench Pro","score":56.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/stepfun-ai/Step-3.7-Flash","measuredAt":"2026-05-23","scope":"Registered model evaluation · Model Card"},{"benchmark":"hle · hle","score":48.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/stepfun-ai/Step-3.7-Flash","measuredAt":"2026-05-23","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"e4da4983-d8f3-41b2-b60b-48fab0eadc75","title":"Agent Memory Architecture Guide","href":"/intel/e4da4983-d8f3-41b2-b60b-48fab0eadc75","hook":"A technical progression from a Python list to graph-vector hybrid memory, with the tradeoffs at each step. The reference to reach for when \"just stuff it in …"},{"id":"c7c0bf34-6301-4f65-9bad-84e1821a7eec","title":"Prompt caching in LLMs, clearly explained","href":"/intel/c7c0bf34-6301-4f65-9bad-84e1821a7eec","hook":"A concrete teardown of how Claude hits a 92% cache hit-rate, and why it matters: every agent step resends the whole history, so caching is the line between a…"},{"id":"0e02d304-1b0c-470b-bb55-1cf6b544d3d1","title":"GPT-5","href":"/intel/0e02d304-1b0c-470b-bb55-1cf6b544d3d1","hook":"GPT-5's default toward autonomous, multi-step action on vague prompts changes how you scope tasks for it — Mollick's hands-on examples show where that proact…"}]}]}]},{"id":"thenlper","name":"Thenlper","domain":null,"major":false,"modelCount":2,"series":[{"id":"gte","label":"GTE","models":[{"id":"thenlper/gte-large-20251117","name":"GTE-Large","releaseLabel":"Large","generationId":"releases","generationLabel":"Releases","variantLabel":"Large","createdAt":"2025-11-18T02:40:55.000Z","kind":"embedding","contextLength":512,"inputModalities":["text"],"outputModalities":["embeddings"],"promptPerMillion":0.01,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"The gte-large embedding model converts English sentences, paragraphs and moderate-length documents into a 1024-dimensional dense vector space, delivering high-quality semantic embeddings optimized for information retrieval, semantic textual similarity, reranking and...","huggingFaceId":"thenlper/gte-large","routes":[{"provider":"OpenRouter","modelId":"thenlper/gte-large","url":"https://openrouter.ai/thenlper/gte-large","free":false,"batch":false}],"benchmarks":[{"benchmark":"MTEB English retrieval","score":53.3,"unit":"%","sourceName":"MTEB","sourceUrl":"https://leaderboard.mteb.org/","measuredAt":"2026-08-27","scope":"Mean retrieval score · leaderboard rank 68"}],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[]},{"id":"thenlper/gte-base-20251117","name":"GTE-Base","releaseLabel":"Base","generationId":"releases","generationLabel":"Releases","variantLabel":"Base","createdAt":"2025-11-18T02:43:40.000Z","kind":"embedding","contextLength":512,"inputModalities":["text"],"outputModalities":["embeddings"],"promptPerMillion":0.005,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"The gte-base embedding model encodes English sentences and paragraphs into a 768-dimensional dense vector space, delivering efficient and effective semantic embeddings optimized for textual similarity, semantic search, and clustering applications.","huggingFaceId":"thenlper/gte-base","routes":[{"provider":"OpenRouter","modelId":"thenlper/gte-base","url":"https://openrouter.ai/thenlper/gte-base","free":false,"batch":false}],"benchmarks":[{"benchmark":"MTEB English retrieval","score":51.9,"unit":"%","sourceName":"MTEB","sourceUrl":"https://leaderboard.mteb.org/","measuredAt":"2026-08-27","scope":"Mean retrieval score · leaderboard rank 81"}],"benchmarkLinks":[{"label":"MTEB","url":"https://leaderboard.mteb.org/","note":"Retrieval, classification, clustering, and multilingual embedding tasks"}],"intel":[]}]}]},{"id":"thinkingmachines","name":"Thinking Machines","domain":"thinkingmachines.ai","major":false,"modelCount":2,"series":[{"id":"inkling","label":"Inkling","models":[{"id":"thinkingmachines/inkling-20260715","name":"Inkling","releaseLabel":"Inkling","generationId":"releases","generationLabel":"Releases","variantLabel":"Inkling","createdAt":"2026-07-17T22:05:56.000Z","kind":"multimodal","contextLength":1048576,"inputModalities":["text","image","audio"],"outputModalities":["text"],"promptPerMillion":0.95,"completionPerMillion":4.05,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Inkling is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 41B active parameters out of 975B total. It is designed for general-purpose reasoning, coding, agentic and tool-use systems,...","huggingFaceId":"thinkingmachines/Inkling","routes":[{"provider":"OpenRouter","modelId":"thinkingmachines/inkling","url":"https://openrouter.ai/thinkingmachines/inkling","free":false,"batch":false},{"provider":"OpenRouter","modelId":"thinkingmachines/inkling:batch","url":"https://openrouter.ai/thinkingmachines/inkling:batch","free":false,"batch":true},{"provider":"OpenRouter","modelId":"thinkingmachines/inkling:free","url":"https://openrouter.ai/thinkingmachines/inkling:free","free":true,"batch":false}],"benchmarks":[{"benchmark":"LiveBench overall","score":72.9,"unit":"%","sourceName":"LiveBench","sourceUrl":"https://raw.githubusercontent.com/LiveBench/new-livebench/main/public/table_2026_06_25.csv","measuredAt":"2026-06-25","scope":"Mean across that public release's task columns"},{"benchmark":"AA Intelligence Index","score":42.3,"unit":"","sourceName":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/inkling","measuredAt":"2026-08-27","scope":"Current public index; highest tested effort configuration"},{"benchmark":"LMArena overall","score":1439,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 63 · 18,488 votes"},{"benchmark":"aime_2026 · MathArena/aime 2026","score":97.1,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/thinkingmachines/Inkling","measuredAt":"2026-07-19","scope":"Registered model evaluation · Model Card"},{"benchmark":"gpqa · diamond","score":87.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/thinkingmachines/Inkling","measuredAt":"2026-07-19","scope":"Registered model evaluation · Model Card"},{"benchmark":"hle · hle","score":46,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/thinkingmachines/Inkling","measuredAt":"2026-07-19","scope":"Registered model evaluation · Model Card"},{"benchmark":"MMMU_Pro · mmmu pro standard 10 options","score":73.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/thinkingmachines/Inkling","measuredAt":"2026-07-19","scope":"Registered model evaluation · Model Card"},{"benchmark":"SWE-bench_Pro · SWE Bench Pro","score":54.3,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/thinkingmachines/Inkling","measuredAt":"2026-07-19","scope":"Registered model evaluation · Model Card"},{"benchmark":"SWE-bench_Verified · swe bench % resolved","score":77.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/thinkingmachines/Inkling","measuredAt":"2026-07-19","scope":"Registered model evaluation · Model Card"},{"benchmark":"terminal-bench-2.1 · terminalbench 2 1","score":63.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/thinkingmachines/Inkling","measuredAt":"2026-07-15","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"0e7e9219-5fd3-4c02-9bac-305501b53419","title":"We ran Thinking Machines Inkling through our private coding-agent bench. Same ha","href":"/intel/0e7e9219-5fd3-4c02-9bac-305501b53419","hook":"Independent benchmark data on Thinking Machines' Inkling across a 13-bug repair harness — a reality check on frontier coding-agent claims."}]},{"id":"thinkingmachines/inkling-small-20260730","name":"Inkling Small","releaseLabel":"Small","generationId":"releases","generationLabel":"Releases","variantLabel":"Small","createdAt":"2026-07-30T20:25:17.000Z","kind":"multimodal","contextLength":1048576,"inputModalities":["text","image","audio"],"outputModalities":["text"],"promptPerMillion":0.44999999999999996,"completionPerMillion":1.2,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Inkling Small is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 12B active parameters out of 276B total. It is positioned as the smaller, more efficient member of...","huggingFaceId":"thinkingmachines/Inkling-Small","routes":[{"provider":"OpenRouter","modelId":"thinkingmachines/inkling-small","url":"https://openrouter.ai/thinkingmachines/inkling-small","free":false,"batch":false},{"provider":"OpenRouter","modelId":"thinkingmachines/inkling-small:free","url":"https://openrouter.ai/thinkingmachines/inkling-small:free","free":true,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1412,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 122 · 11,455 votes"},{"benchmark":"aime_2026 · MathArena/aime 2026","score":95.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/thinkingmachines/Inkling-Small","measuredAt":"2026-07-31","scope":"Registered model evaluation · Model Card"},{"benchmark":"gpqa · diamond","score":89.5,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/thinkingmachines/Inkling-Small","measuredAt":"2026-07-31","scope":"Registered model evaluation · Model Card"},{"benchmark":"hle · hle","score":31.6,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/thinkingmachines/Inkling-Small","measuredAt":"2026-07-31","scope":"Registered model evaluation · Model Card"},{"benchmark":"hmmt_feb_2026 · MathArena/hmmt feb 2026","score":90.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/thinkingmachines/Inkling-Small","measuredAt":"2026-07-31","scope":"Registered model evaluation · Model Card"},{"benchmark":"MMMU_Pro · mmmu pro standard 10 options","score":74,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/thinkingmachines/Inkling-Small","measuredAt":"2026-07-31","scope":"Registered model evaluation · Model Card"},{"benchmark":"SWE-bench_Pro · SWE Bench Pro","score":55.9,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/thinkingmachines/Inkling-Small","measuredAt":"2026-07-31","scope":"Registered model evaluation · Model Card"},{"benchmark":"SWE-bench_Verified · swe bench % resolved","score":80.2,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/thinkingmachines/Inkling-Small","measuredAt":"2026-07-31","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"0e7e9219-5fd3-4c02-9bac-305501b53419","title":"We ran Thinking Machines Inkling through our private coding-agent bench. Same ha","href":"/intel/0e7e9219-5fd3-4c02-9bac-305501b53419","hook":"Independent benchmark data on Thinking Machines' Inkling across a 13-bug repair harness — a reality check on frontier coding-agent claims."}]}]}]},{"id":"upstage","name":"Upstage","domain":"upstage.ai","major":false,"modelCount":2,"series":[{"id":"solar","label":"Solar","models":[{"id":"upstage/solar-pro-3","name":"Solar Pro 3","releaseLabel":"Pro 3","generationId":"3","generationLabel":"3","variantLabel":"Pro","createdAt":"2026-01-27T02:33:20.000Z","kind":"language","contextLength":131072,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.15,"completionPerMillion":0.6,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Solar Pro 3 is Upstage's powerful Mixture-of-Experts (MoE) language model. With 102B total parameters and 12B active parameters per forward pass, it delivers exceptional performance while maintaining computational efficiency. Optimized...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"upstage/solar-pro-3","url":"https://openrouter.ai/upstage/solar-pro-3","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]},{"id":"upstage/solar-pro4-20260810","name":"Solar Pro 4","releaseLabel":"Pro 4","generationId":"releases","generationLabel":"Releases","variantLabel":"Pro 4","createdAt":"2026-08-10T14:20:36.000Z","kind":"language","contextLength":524288,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.03,"completionPerMillion":0.12,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Solar Pro 4 is Upstage's cost-efficient large language model, featuring a 524K context window. It is built for long-horizon tasks and agentic workflows, with strong capabilities in office productivity, document-intensive...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"upstage/solar-pro4","url":"https://openrouter.ai/upstage/solar-pro4","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]}]},{"id":"allenai","name":"Ai2","domain":"allenai.org","major":false,"modelCount":1,"series":[{"id":"olmo","label":"Olmo","models":[{"id":"allenai/olmo-3-32b-think-20251121","name":"Olmo 3 32B Think","releaseLabel":"3 32B Think","generationId":"3","generationLabel":"3","variantLabel":"32B Think","createdAt":"2025-11-21T20:51:16.000Z","kind":"language","contextLength":65536,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.15,"completionPerMillion":0.5,"imageOutput":null,"supportsTools":false,"supportsReasoning":true,"description":"Olmo 3 32B Think is a large-scale, 32-billion-parameter model purpose-built for deep reasoning, complex logic chains and advanced instruction-following scenarios. Its capacity enables strong performance on demanding evaluation tasks and...","huggingFaceId":"allenai/Olmo-3-32B-Think","routes":[{"provider":"OpenRouter","modelId":"allenai/olmo-3-32b-think","url":"https://openrouter.ai/allenai/olmo-3-32b-think","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1299,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 229 · 5,968 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[{"id":"304106f1-b55b-4812-b98c-2ae6844c6b7f","title":"Interconnects: Frontier Post-Training Recipe Review","href":"/intel/304106f1-b55b-4812-b98c-2ae6844c6b7f","hook":"It gives ML practitioners a clear, comparative map of how post-training recipes differ across DeepSeek, Llama, Tülü, OLMo, Nemotron, Kimi, and GLM, and expla…"}]}]}]},{"id":"anthracite-org","name":"Anthracite ORG","domain":null,"major":false,"modelCount":1,"series":[{"id":"magnum","label":"Magnum","models":[{"id":"anthracite-org/magnum-v4-72b","name":"Magnum v4 72B","releaseLabel":"v4 72B","generationId":"v4","generationLabel":"V4","variantLabel":"72B","createdAt":"2024-10-22T00:00:00.000Z","kind":"language","contextLength":32768,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":3,"completionPerMillion":5,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"This is a series of models designed to replicate the prose quality of the Claude 3 models, specifically Sonnet(https://openrouter.ai/anthropic/claude-3.5-sonnet) and Opus(https://openrouter.ai/anthropic/claude-3-opus). The model is fine-tuned on top of [Qwen2.5 72B](https://openrouter.ai/qwen/qwen-2.5-72b-instruct).","huggingFaceId":"anthracite-org/magnum-v4-72b","routes":[{"provider":"OpenRouter","modelId":"anthracite-org/magnum-v4-72b","url":"https://openrouter.ai/anthracite-org/magnum-v4-72b","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]}]},{"id":"baidu","name":"Baidu","domain":"baidu.com","major":false,"modelCount":1,"series":[{"id":"ernie","label":"Ernie","models":[{"id":"baidu/ernie-4.5-vl-424b-a47b","name":"ERNIE 4.5 VL 424B A47B","releaseLabel":"4.5 VL 424B A47B","generationId":"4.5","generationLabel":"4.5","variantLabel":"VL 424B A47B","createdAt":"2025-06-30T16:28:23.000Z","kind":"multimodal","contextLength":123000,"inputModalities":["image","text"],"outputModalities":["text"],"promptPerMillion":0.42,"completionPerMillion":1.25,"imageOutput":null,"supportsTools":false,"supportsReasoning":true,"description":"ERNIE-4.5-VL-424B-A47B is a multimodal Mixture-of-Experts (MoE) model from Baidu’s ERNIE 4.5 series, featuring 424B total parameters with 47B active per token. It is trained jointly on text and image data...","huggingFaceId":"baidu/ERNIE-4.5-VL-424B-A47B-PT","routes":[{"provider":"OpenRouter","modelId":"baidu/ernie-4.5-vl-424b-a47b","url":"https://openrouter.ai/baidu/ernie-4.5-vl-424b-a47b","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]}]},{"id":"canopylabs","name":"Canopy Labs","domain":"canopylabs.ai","major":false,"modelCount":1,"series":[{"id":"orpheus","label":"Orpheus","models":[{"id":"canopylabs/orpheus-3b-0.1-ft","name":"Orpheus 3B","releaseLabel":"3B","generationId":"3b","generationLabel":"3B","variantLabel":"0.1 Ft","createdAt":"2026-04-23T22:26:08.000Z","kind":"audio","contextLength":4096,"inputModalities":["text"],"outputModalities":["speech"],"promptPerMillion":7,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Orpheus 3B is an English text-to-speech model from Canopy Labs, fine-tuned for natural prosody and expressive delivery. It offers 7 preset voices and is suited for narration, voice assistants, and...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"canopylabs/orpheus-3b-0.1-ft","url":"https://openrouter.ai/canopylabs/orpheus-3b-0.1-ft","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[]}]}]},{"id":"cognitivecomputations","name":"Cognitivecomputations","domain":null,"major":false,"modelCount":1,"series":[{"id":"dolphin","label":"Dolphin","models":[{"id":"cognitivecomputations/uncensored","name":"Uncensored","releaseLabel":"Uncensored","generationId":"24b","generationLabel":"24B","variantLabel":"Venice Edition","createdAt":"2025-07-09T21:02:46.000Z","kind":"language","contextLength":128000,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.19999999999999998,"completionPerMillion":0.8999999999999999,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Venice Uncensored Dolphin Mistral 24B Venice Edition is a fine-tuned variant of Mistral-Small-24B-Instruct-2501, developed by dphn.ai in collaboration with Venice.ai. This model is designed as an “uncensored” instruct-tuned LLM, preserving...","huggingFaceId":"cognitivecomputations/Dolphin-Mistral-24B-Venice-Edition","routes":[{"provider":"OpenRouter","modelId":"cognitivecomputations/dolphin-mistral-24b-venice-edition","url":"https://openrouter.ai/cognitivecomputations/dolphin-mistral-24b-venice-edition","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]}]},{"id":"dots-studio","name":"Dots Studio","domain":null,"major":false,"modelCount":1,"series":[{"id":"dots","label":"Dots","models":[{"id":"dots-studio/dots-3-note-preview-20260813","name":"Dots3-Note Preview","releaseLabel":"3-Note Preview","generationId":"3","generationLabel":"3","variantLabel":"Note Preview","createdAt":"2026-08-14T04:06:01.000Z","kind":"multimodal","contextLength":512000,"inputModalities":["text","image"],"outputModalities":["text"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Dots3-Note Preview is an open-weight mixture-of-experts model from Dots Studio, with 16B active parameters out of 280B total. It is the lightest model in the Dots 3 family and is...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"dots-studio/dots-3-note-preview:free","url":"https://openrouter.ai/dots-studio/dots-3-note-preview:free","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]}]},{"id":"gryphe","name":"Gryphe","domain":null,"major":false,"modelCount":1,"series":[{"id":"mythomax","label":"Mythomax","models":[{"id":"gryphe/mythomax-l2-13b","name":"MythoMax 13B","releaseLabel":"13B","generationId":"13b","generationLabel":"13B","variantLabel":"Base","createdAt":"2023-07-02T00:00:00.000Z","kind":"language","contextLength":8192,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.06,"completionPerMillion":0.06,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"One of the highest performing and most popular fine-tunes of Llama 2 13B, with rich descriptions and roleplay. #merge","huggingFaceId":"Gryphe/MythoMax-L2-13b","routes":[{"provider":"OpenRouter","modelId":"gryphe/mythomax-l2-13b","url":"https://openrouter.ai/gryphe/mythomax-l2-13b","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]}]},{"id":"hexgrad","name":"Hexgrad","domain":null,"major":false,"modelCount":1,"series":[{"id":"kokoro","label":"Kokoro","models":[{"id":"hexgrad/kokoro-82m","name":"Kokoro 82M","releaseLabel":"82M","generationId":"82m","generationLabel":"82m","variantLabel":"Base","createdAt":"2026-04-23T22:26:07.000Z","kind":"audio","contextLength":4096,"inputModalities":["text"],"outputModalities":["speech"],"promptPerMillion":0.62,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Kokoro 82M is a lightweight, open-weight text-to-speech model from hexgrad. It converts text to speech across 8 languages (American and British English, Spanish, French, Hindi, Italian, Japanese, Portuguese, and Chinese)...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"hexgrad/kokoro-82m","url":"https://openrouter.ai/hexgrad/kokoro-82m","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[]}]}]},{"id":"heygen","name":"Heygen","domain":null,"major":false,"modelCount":1,"series":[{"id":"avatar","label":"Avatar","models":[{"id":"heygen/avatar-iv-20260625","name":"Avatar IV","releaseLabel":"IV","generationId":"releases","generationLabel":"Releases","variantLabel":"Iv","createdAt":"2026-08-24T18:22:08.000Z","kind":"video","contextLength":0,"inputModalities":["text","image","audio"],"outputModalities":["video"],"promptPerMillion":0,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"HeyGen: Avatar IV is an image-to-video model that animates a single photo into an expressive, lip-synced talking-head video. Rather than only matching mouth shapes to words, it interprets the vocal...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"heygen/avatar-iv","url":"https://openrouter.ai/heygen/avatar-iv","free":true,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis Video Arena","url":"https://artificialanalysis.ai/video/leaderboard/text-to-video","note":"Blind preference ranking for video generation"}],"intel":[]}]}]},{"id":"inception","name":"Inception","domain":"inceptionlabs.ai","major":false,"modelCount":1,"series":[{"id":"mercury","label":"Mercury","models":[{"id":"inception/mercury-2-20260304","name":"Mercury 2","releaseLabel":"2","generationId":"2","generationLabel":"2","variantLabel":"Base","createdAt":"2026-03-04T14:57:55.000Z","kind":"language","contextLength":128000,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.25,"completionPerMillion":0.75,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"Mercury 2 is an extremely fast reasoning LLM, and the first reasoning diffusion LLM (dLLM). Instead of generating tokens sequentially, Mercury 2 produces and refines multiple tokens in parallel, achieving...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"inception/mercury-2","url":"https://openrouter.ai/inception/mercury-2","free":false,"batch":false}],"benchmarks":[{"benchmark":"LMArena overall","score":1358,"unit":"","sourceName":"LMArena","sourceUrl":"https://arena.ai/leaderboard/text","measuredAt":"2026-08-21","scope":"Human preference Elo · rank 180 · 3,122 votes"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]}]},{"id":"mancer","name":"Mancer","domain":null,"major":false,"modelCount":1,"series":[{"id":"weaver","label":"Weaver","models":[{"id":"mancer/weaver","name":"Weaver (alpha)","releaseLabel":"(alpha)","generationId":"releases","generationLabel":"Releases","variantLabel":"(alpha)","createdAt":"2023-08-02T00:00:00.000Z","kind":"language","contextLength":8000,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.5,"completionPerMillion":0.75,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"An attempt to recreate Claude-style verbosity, but don't expect the same level of coherence or memory. Meant for use in roleplay/narrative situations.","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"mancer/weaver","url":"https://openrouter.ai/mancer/weaver","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]}]},{"id":"meituan","name":"Meituan","domain":"meituan.com","major":false,"modelCount":1,"series":[{"id":"longcat","label":"Longcat","models":[{"id":"meituan/longcat-2.0-20260720","name":"LongCat 2.0","releaseLabel":"2.0","generationId":"2.0","generationLabel":"2.0","variantLabel":"Base","createdAt":"2026-07-20T13:37:38.000Z","kind":"language","contextLength":1048756,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.3,"completionPerMillion":1.2,"imageOutput":null,"supportsTools":true,"supportsReasoning":true,"description":"LongCat 2.0 is a sparse mixture-of-experts language model from Meituan, with 48B active parameters out of 1.6T total. It is suited for coding, repository-level changes, long-horizon problem solving, and agentic...","huggingFaceId":"meituan-longcat/LongCat-2.0","routes":[{"provider":"OpenRouter","modelId":"meituan/longcat-2.0","url":"https://openrouter.ai/meituan/longcat-2.0","free":false,"batch":false}],"benchmarks":[{"benchmark":"terminal-bench-2.1 · terminalbench 2 1","score":70.8,"unit":"%","sourceName":"Hugging Face eval result","sourceUrl":"https://huggingface.co/meituan-longcat/LongCat-2.0","measuredAt":"2026-07-08","scope":"Registered model evaluation · Model Card"}],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]}]},{"id":"perceptron","name":"Perceptron","domain":null,"major":false,"modelCount":1,"series":[{"id":"perceptron","label":"Perceptron","models":[{"id":"perceptron/perceptron-mk1-20260512","name":"Perceptron Mk1","releaseLabel":"Mk1","generationId":"releases","generationLabel":"Releases","variantLabel":"Mk1","createdAt":"2026-05-12T14:43:49.000Z","kind":"multimodal","contextLength":32768,"inputModalities":["text","image","video"],"outputModalities":["text"],"promptPerMillion":0.15,"completionPerMillion":1.5,"imageOutput":null,"supportsTools":false,"supportsReasoning":true,"description":"Perceptron Mk1 (Mark One) is Perceptron's highest-quality vision-language model for video and embodied reasoning.** It accepts image and video inputs paired with natural language queries, and produces detailed visual understanding...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"perceptron/perceptron-mk1","url":"https://openrouter.ai/perceptron/perceptron-mk1","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]}]},{"id":"sesame","name":"Sesame","domain":"sesame.com","major":false,"modelCount":1,"series":[{"id":"csm","label":"CSM","models":[{"id":"sesame/csm-1b","name":"CSM 1B","releaseLabel":"1B","generationId":"1b","generationLabel":"1B","variantLabel":"Base","createdAt":"2026-04-23T22:26:08.000Z","kind":"audio","contextLength":4096,"inputModalities":["text"],"outputModalities":["speech"],"promptPerMillion":7,"completionPerMillion":0,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"CSM 1B is a conversational speech model from Sesame. It accepts text input and produces English speech output, with voice options spanning conversational and read-speech styles. At 1B parameters, it...","huggingFaceId":null,"routes":[{"provider":"OpenRouter","modelId":"sesame/csm-1b","url":"https://openrouter.ai/sesame/csm-1b","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"Artificial Analysis media benchmarks","url":"https://artificialanalysis.ai/evaluations","note":"Speech and music evaluations where available"}],"intel":[]}]}]},{"id":"undi95","name":"Undi95","domain":null,"major":false,"modelCount":1,"series":[{"id":"remm","label":"Remm","models":[{"id":"undi95/remm-slerp-l2-13b","name":"ReMM SLERP 13B","releaseLabel":"SLERP 13B","generationId":"13b","generationLabel":"13B","variantLabel":"Slerp","createdAt":"2023-07-22T00:00:00.000Z","kind":"language","contextLength":6144,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.44999999999999996,"completionPerMillion":0.65,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"A recreation trial of the original MythoMax-L2-B13 but with updated models. #merge","huggingFaceId":"Undi95/ReMM-SLERP-L2-13B","routes":[{"provider":"OpenRouter","modelId":"undi95/remm-slerp-l2-13b","url":"https://openrouter.ai/undi95/remm-slerp-l2-13b","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]}]},{"id":"writer","name":"Writer","domain":"writer.com","major":false,"modelCount":1,"series":[{"id":"palmyra","label":"Palmyra","models":[{"id":"writer/palmyra-x5-20250428","name":"Palmyra X5","releaseLabel":"X5","generationId":"releases","generationLabel":"Releases","variantLabel":"X5","createdAt":"2026-01-21T13:57:03.000Z","kind":"language","contextLength":1040000,"inputModalities":["text"],"outputModalities":["text"],"promptPerMillion":0.6,"completionPerMillion":6,"imageOutput":null,"supportsTools":false,"supportsReasoning":false,"description":"Palmyra X5 is Writer's most advanced model, purpose-built for building and scaling AI agents across the enterprise. It delivers industry-leading speed and efficiency on context windows up to 1 million...","huggingFaceId":"","routes":[{"provider":"OpenRouter","modelId":"writer/palmyra-x5","url":"https://openrouter.ai/writer/palmyra-x5","free":false,"batch":false}],"benchmarks":[],"benchmarkLinks":[{"label":"LiveBench","url":"https://livebench.ai/","note":"Contamination-resistant language and coding tasks"},{"label":"Artificial Analysis","url":"https://artificialanalysis.ai/models","note":"Independent intelligence, coding, agentic, speed, and price data"}],"intel":[]}]}]}]}