{"data":[{"entity_id":"alibaba-happyhorse-1-0","fields":{"name":"HappyHorse 1.0","vendor":"Alibaba","released_at":"2026-04-07","modality":"video","access":"closed","announcement_url":"https://www.cnbc.com/2026/04/10/alibaba-happyhorse-ai-video-model-benchmark-reveal.html"},"evidence":{"name":{"url":"https://www.buildfastwithai.com/blogs/best-ai-models-may-2026-leaderboard","retrieved_at":"2026-05-04T07:13:00Z","title":"Best AI Models: April + May 2026 Leaderboard","excerpt":"HappyHorse 1.0 appeared anonymously on April 7, 2026 on the Artificial Analysis Video Arena, achieved the highest Elo score in AI video history (1389). Alibaba revealed authorship on April 10. 15B parameter model from Alibaba's Future Life Lab / Taotian Group."},"vendor":{"url":"https://www.buildfastwithai.com/blogs/best-ai-models-may-2026-leaderboard","retrieved_at":"2026-05-04T07:13:00Z","title":"Best AI Models: April + May 2026 Leaderboard","excerpt":"HappyHorse 1.0 appeared anonymously on April 7, 2026 on the Artificial Analysis Video Arena, achieved the highest Elo score in AI video history (1389). Alibaba revealed authorship on April 10. 15B parameter model from Alibaba's Future Life Lab / Taotian Group."},"released_at":{"url":"https://www.buildfastwithai.com/blogs/best-ai-models-may-2026-leaderboard","retrieved_at":"2026-05-04T07:13:00Z","title":"Best AI Models: April + May 2026 Leaderboard","excerpt":"HappyHorse 1.0 appeared anonymously on April 7, 2026 on the Artificial Analysis Video Arena, achieved the highest Elo score in AI video history (1389). Alibaba revealed authorship on April 10. 15B parameter model from Alibaba's Future Life Lab / Taotian Group."},"modality":{"url":"https://www.buildfastwithai.com/blogs/best-ai-models-may-2026-leaderboard","retrieved_at":"2026-05-04T07:13:00Z","title":"Best AI Models: April + May 2026 Leaderboard","excerpt":"HappyHorse 1.0 appeared anonymously on April 7, 2026 on the Artificial Analysis Video Arena, achieved the highest Elo score in AI video history (1389). Alibaba revealed authorship on April 10. 15B parameter model from Alibaba's Future Life Lab / Taotian Group."},"access":{"url":"https://www.buildfastwithai.com/blogs/best-ai-models-may-2026-leaderboard","retrieved_at":"2026-05-04T07:13:00Z","title":"Best AI Models: April + May 2026 Leaderboard","excerpt":"HappyHorse 1.0 appeared anonymously on April 7, 2026 on the Artificial Analysis Video Arena, achieved the highest Elo score in AI video history (1389). Alibaba revealed authorship on April 10. 15B parameter model from Alibaba's Future Life Lab / Taotian Group."},"announcement_url":{"url":"https://www.buildfastwithai.com/blogs/best-ai-models-may-2026-leaderboard","retrieved_at":"2026-05-04T07:13:00Z","title":"Best AI Models: April + May 2026 Leaderboard","excerpt":"HappyHorse 1.0 appeared anonymously on April 7, 2026 on the Artificial Analysis Video Arena, achieved the highest Elo score in AI video history (1389). Alibaba revealed authorship on April 10. 15B parameter model from Alibaba's Future Life Lab / Taotian Group."}}},{"entity_id":"anthropic-claude-opus-4-7","fields":{"name":"Claude Opus 4.7","vendor":"Anthropic","released_at":"2026-04-16","modality":"text, code, vision","access":"closed","benchmark_score":64.3,"benchmark_name":"SWE-Bench Pro","announcement_url":"https://www.anthropic.com/news/claude-opus-4-7"},"evidence":{"name":{"url":"https://www.anthropic.com/news/claude-opus-4-7","retrieved_at":"2026-04-30","title":"Introducing Claude Opus 4.7 | Anthropic","excerpt":"Claude Opus 4.7: notable improvement on Opus 4.6 in advanced software engineering. Accepts images up to 2,576 pixels. $5/M input, $25/M output tokens."},"vendor":{"url":"https://www.anthropic.com/news/claude-opus-4-7","retrieved_at":"2026-04-30","title":"Introducing Claude Opus 4.7 | Anthropic","excerpt":"Claude Opus 4.7: notable improvement on Opus 4.6 in advanced software engineering. Accepts images up to 2,576 pixels. $5/M input, $25/M output tokens."},"released_at":{"url":"https://www.anthropic.com/news/claude-opus-4-7","retrieved_at":"2026-04-30","title":"Introducing Claude Opus 4.7 | Anthropic","excerpt":"Claude Opus 4.7: notable improvement on Opus 4.6 in advanced software engineering. Accepts images up to 2,576 pixels. $5/M input, $25/M output tokens."},"modality":{"url":"https://www.anthropic.com/news/claude-opus-4-7","retrieved_at":"2026-04-30","title":"Introducing Claude Opus 4.7 | Anthropic","excerpt":"Claude Opus 4.7: notable improvement on Opus 4.6 in advanced software engineering. Accepts images up to 2,576 pixels. $5/M input, $25/M output tokens."},"access":{"url":"https://www.anthropic.com/news/claude-opus-4-7","retrieved_at":"2026-04-30","title":"Introducing Claude Opus 4.7 | Anthropic","excerpt":"Claude Opus 4.7: notable improvement on Opus 4.6 in advanced software engineering. Accepts images up to 2,576 pixels. $5/M input, $25/M output tokens."},"benchmark_score":{"url":"https://www.buildfastwithai.com/blogs/best-ai-models-may-2026-leaderboard","retrieved_at":"2026-05-03T07:00:00Z","title":"Best AI Models: April + May 2026 Leaderboard (GPT-5.5, Claude Opus 4.7, DeepSeek V4)","excerpt":"Claude Opus 4.7 pushed SWE-bench Pro from 53.4% to 64.3%, reclaiming the coding crown. Claude Opus 4.7 leads SWE-bench Pro at 64.3% for real GitHub issue resolution."},"benchmark_name":{"url":"https://www.buildfastwithai.com/blogs/best-ai-models-may-2026-leaderboard","retrieved_at":"2026-05-03T07:00:00Z","title":"Best AI Models: April + May 2026 Leaderboard (GPT-5.5, Claude Opus 4.7, DeepSeek V4)","excerpt":"Claude Opus 4.7 pushed SWE-bench Pro from 53.4% to 64.3%, reclaiming the coding crown."},"announcement_url":{"url":"https://www.anthropic.com/news/claude-opus-4-7","retrieved_at":"2026-04-30","title":"Introducing Claude Opus 4.7 | Anthropic","excerpt":"Claude Opus 4.7: notable improvement on Opus 4.6 in advanced software engineering. Accepts images up to 2,576 pixels. $5/M input, $25/M output tokens."}}},{"entity_id":"anthropic-claude-sonnet-4-6","fields":{"name":"Claude Sonnet 4.6","vendor":"Anthropic","released_at":"2026-02-17","modality":"text, code, vision","access":"closed","context_window":500000,"benchmark_score":80.8,"benchmark_name":"SWE-Bench Verified","announcement_url":"https://www.anthropic.com/news/claude-sonnet-4-6"},"evidence":{"name":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"Claude Sonnet 4.6 released 2026-02-17 by Anthropic. ~500B, 500K context, 80.8% SWE-bench Verified. Near-Opus quality at a fraction of the cost, with Agent Teams orchestration."},"vendor":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"Claude Sonnet 4.6 released 2026-02-17 by Anthropic. ~500B, 500K context, 80.8% SWE-bench Verified. Near-Opus quality at a fraction of the cost, with Agent Teams orchestration."},"released_at":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"Claude Sonnet 4.6 released 2026-02-17 by Anthropic. ~500B, 500K context, 80.8% SWE-bench Verified. Near-Opus quality at a fraction of the cost, with Agent Teams orchestration."},"modality":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"Claude Sonnet 4.6 released 2026-02-17 by Anthropic. ~500B, 500K context, 80.8% SWE-bench Verified. Near-Opus quality at a fraction of the cost, with Agent Teams orchestration."},"access":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"Claude Sonnet 4.6 released 2026-02-17 by Anthropic. ~500B, 500K context, 80.8% SWE-bench Verified. Near-Opus quality at a fraction of the cost, with Agent Teams orchestration."},"context_window":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"Claude Sonnet 4.6 released 2026-02-17 by Anthropic. ~500B, 500K context, 80.8% SWE-bench Verified. Near-Opus quality at a fraction of the cost, with Agent Teams orchestration."},"benchmark_score":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"Claude Sonnet 4.6 released 2026-02-17 by Anthropic. ~500B, 500K context, 80.8% SWE-bench Verified. Near-Opus quality at a fraction of the cost, with Agent Teams orchestration."},"benchmark_name":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"Claude Sonnet 4.6 released 2026-02-17 by Anthropic. ~500B, 500K context, 80.8% SWE-bench Verified. Near-Opus quality at a fraction of the cost, with Agent Teams orchestration."},"announcement_url":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"Claude Sonnet 4.6 released 2026-02-17 by Anthropic. ~500B, 500K context, 80.8% SWE-bench Verified. Near-Opus quality at a fraction of the cost, with Agent Teams orchestration."}}},{"entity_id":"arcee-trinity-large-thinking","fields":{"name":"Trinity-Large-Thinking","vendor":"Arcee AI","released_at":"2026-04-01","modality":"text, code","access":"open-weights","parameters":"399B total / 13B active","benchmark_score":63.2,"benchmark_name":"SWE-Bench Verified","announcement_url":"https://www.arcee.ai/blog/trinity-large-thinking"},"evidence":{"name":{"url":"https://www.arcee.ai/blog/trinity-large-thinking","retrieved_at":"2026-05-05T06:55:00Z","title":"Trinity-Large-Thinking: Scaling an Open Source Frontier Agent","excerpt":"Trinity-Large-Thinking is live. A frontier open reasoning model for complex, long-horizon agents and multi-turn tool calling released under Apache 2.0. 399B total parameters, 13B active (4-of-256 experts). SWE-Bench Verified 63.2%."},"vendor":{"url":"https://www.arcee.ai/blog/trinity-large-thinking","retrieved_at":"2026-05-05T06:55:00Z","title":"Trinity-Large-Thinking: Scaling an Open Source Frontier Agent","excerpt":"Trinity-Large-Thinking is live. A frontier open reasoning model for complex, long-horizon agents and multi-turn tool calling released under Apache 2.0. 399B total parameters, 13B active (4-of-256 experts). SWE-Bench Verified 63.2%."},"released_at":{"url":"https://www.arcee.ai/blog/trinity-large-thinking","retrieved_at":"2026-05-05T06:55:00Z","title":"Trinity-Large-Thinking: Scaling an Open Source Frontier Agent","excerpt":"Trinity-Large-Thinking is live. A frontier open reasoning model for complex, long-horizon agents and multi-turn tool calling released under Apache 2.0. 399B total parameters, 13B active (4-of-256 experts). SWE-Bench Verified 63.2%."},"modality":{"url":"https://www.arcee.ai/blog/trinity-large-thinking","retrieved_at":"2026-05-05T06:55:00Z","title":"Trinity-Large-Thinking: Scaling an Open Source Frontier Agent","excerpt":"Trinity-Large-Thinking is live. A frontier open reasoning model for complex, long-horizon agents and multi-turn tool calling released under Apache 2.0. 399B total parameters, 13B active (4-of-256 experts). SWE-Bench Verified 63.2%."},"access":{"url":"https://www.arcee.ai/blog/trinity-large-thinking","retrieved_at":"2026-05-05T06:55:00Z","title":"Trinity-Large-Thinking: Scaling an Open Source Frontier Agent","excerpt":"Trinity-Large-Thinking is live. A frontier open reasoning model for complex, long-horizon agents and multi-turn tool calling released under Apache 2.0. 399B total parameters, 13B active (4-of-256 experts). SWE-Bench Verified 63.2%."},"parameters":{"url":"https://www.arcee.ai/blog/trinity-large-thinking","retrieved_at":"2026-05-05T06:55:00Z","title":"Trinity-Large-Thinking: Scaling an Open Source Frontier Agent","excerpt":"Trinity-Large-Thinking is live. A frontier open reasoning model for complex, long-horizon agents and multi-turn tool calling released under Apache 2.0. 399B total parameters, 13B active (4-of-256 experts). SWE-Bench Verified 63.2%."},"benchmark_score":{"url":"https://www.arcee.ai/blog/trinity-large-thinking","retrieved_at":"2026-05-05T06:55:00Z","title":"Trinity-Large-Thinking: Scaling an Open Source Frontier Agent","excerpt":"Trinity-Large-Thinking is live. A frontier open reasoning model for complex, long-horizon agents and multi-turn tool calling released under Apache 2.0. 399B total parameters, 13B active (4-of-256 experts). SWE-Bench Verified 63.2%."},"benchmark_name":{"url":"https://www.arcee.ai/blog/trinity-large-thinking","retrieved_at":"2026-05-05T06:55:00Z","title":"Trinity-Large-Thinking: Scaling an Open Source Frontier Agent","excerpt":"Trinity-Large-Thinking is live. A frontier open reasoning model for complex, long-horizon agents and multi-turn tool calling released under Apache 2.0. 399B total parameters, 13B active (4-of-256 experts). SWE-Bench Verified 63.2%."},"announcement_url":{"url":"https://www.arcee.ai/blog/trinity-large-thinking","retrieved_at":"2026-05-05T06:55:00Z","title":"Trinity-Large-Thinking: Scaling an Open Source Frontier Agent","excerpt":"Trinity-Large-Thinking is live. A frontier open reasoning model for complex, long-horizon agents and multi-turn tool calling released under Apache 2.0. 399B total parameters, 13B active (4-of-256 experts). SWE-Bench Verified 63.2%."}}},{"entity_id":"bytedance-seedance-2-0","fields":{"name":"Seedance 2.0","vendor":"ByteDance","released_at":"2026-02-10","modality":"video","access":"closed","announcement_url":"https://volcengine.com/product/seedance"},"evidence":{"name":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"Seedance 2.0 released 2026-02-10 by ByteDance. Video generation model with synchronized audio and video in a single pass, 8+ languages for lip-sync."},"vendor":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"Seedance 2.0 released 2026-02-10 by ByteDance. Video generation model with synchronized audio and video in a single pass, 8+ languages for lip-sync."},"released_at":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"Seedance 2.0 released 2026-02-10 by ByteDance. Video generation model with synchronized audio and video in a single pass, 8+ languages for lip-sync."},"modality":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"Seedance 2.0 released 2026-02-10 by ByteDance. Video generation model with synchronized audio and video in a single pass, 8+ languages for lip-sync."},"access":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"Seedance 2.0 released 2026-02-10 by ByteDance. Video generation model with synchronized audio and video in a single pass, 8+ languages for lip-sync."},"announcement_url":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"Seedance 2.0 released 2026-02-10 by ByteDance. Video generation model with synchronized audio and video in a single pass, 8+ languages for lip-sync."}}},{"entity_id":"deepseek-v3-2","fields":{"name":"DeepSeek V3.2","vendor":"DeepSeek","released_at":"2026-02-12","modality":"text, code","access":"open-weights","parameters":"671B total / 37B active","context_window":1000000,"announcement_url":"https://huggingface.co/deepseek-ai/DeepSeek-V3-2"},"evidence":{"name":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"DeepSeek V3.2 released 2026-02-12. 671B MoE (37B active), 1M+ token context window (10x expansion from V3). Open weights under DeepSeek License."},"vendor":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"DeepSeek V3.2 released 2026-02-12. 671B MoE (37B active), 1M+ token context window (10x expansion from V3). Open weights under DeepSeek License."},"released_at":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"DeepSeek V3.2 released 2026-02-12. 671B MoE (37B active), 1M+ token context window (10x expansion from V3). Open weights under DeepSeek License."},"modality":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"DeepSeek V3.2 released 2026-02-12. 671B MoE (37B active), 1M+ token context window (10x expansion from V3). Open weights under DeepSeek License."},"access":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"DeepSeek V3.2 released 2026-02-12. 671B MoE (37B active), 1M+ token context window (10x expansion from V3). Open weights under DeepSeek License."},"parameters":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"DeepSeek V3.2 released 2026-02-12. 671B MoE (37B active), 1M+ token context window (10x expansion from V3). Open weights under DeepSeek License."},"context_window":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"DeepSeek V3.2 released 2026-02-12. 671B MoE (37B active), 1M+ token context window (10x expansion from V3). Open weights under DeepSeek License."},"announcement_url":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"DeepSeek V3.2 released 2026-02-12. 671B MoE (37B active), 1M+ token context window (10x expansion from V3). Open weights under DeepSeek License."}}},{"entity_id":"deepseek-v4-flash","fields":{"name":"DeepSeek-V4-Flash","vendor":"DeepSeek","released_at":"2026-04-24","modality":"text, code","access":"open-weights","parameters":"284B total / 13B active","context_window":1000000,"announcement_url":"https://api-docs.deepseek.com/news/news260424"},"evidence":{"name":{"url":"https://api-docs.deepseek.com/news/news260424","retrieved_at":"2026-05-01T07:00:00Z","title":"DeepSeek V4 Preview Release","excerpt":"DeepSeek-V4-Flash: 284B total / 13B active params. Fast, efficient, and economical choice. 1M context length."},"vendor":{"url":"https://api-docs.deepseek.com/news/news260424","retrieved_at":"2026-05-01T07:00:00Z","title":"DeepSeek V4 Preview Release","excerpt":"DeepSeek-V4-Flash: 284B total / 13B active params. Fast, efficient, and economical choice. 1M context length."},"released_at":{"url":"https://api-docs.deepseek.com/news/news260424","retrieved_at":"2026-05-01T07:00:00Z","title":"DeepSeek V4 Preview Release","excerpt":"DeepSeek-V4-Flash: 284B total / 13B active params. Fast, efficient, and economical choice. 1M context length."},"modality":{"url":"https://api-docs.deepseek.com/news/news260424","retrieved_at":"2026-05-01T07:00:00Z","title":"DeepSeek V4 Preview Release","excerpt":"DeepSeek-V4-Flash: 284B total / 13B active params. Fast, efficient, and economical choice. 1M context length."},"access":{"url":"https://api-docs.deepseek.com/news/news260424","retrieved_at":"2026-05-01T07:00:00Z","title":"DeepSeek V4 Preview Release","excerpt":"DeepSeek-V4-Flash: 284B total / 13B active params. Fast, efficient, and economical choice. 1M context length."},"parameters":{"url":"https://api-docs.deepseek.com/news/news260424","retrieved_at":"2026-05-01T07:00:00Z","title":"DeepSeek V4 Preview Release","excerpt":"DeepSeek-V4-Flash: 284B total / 13B active params. Fast, efficient, and economical choice. 1M context length."},"context_window":{"url":"https://api-docs.deepseek.com/news/news260424","retrieved_at":"2026-05-01T07:00:00Z","title":"DeepSeek V4 Preview Release","excerpt":"DeepSeek-V4-Flash: 284B total / 13B active params. Fast, efficient, and economical choice. 1M context length."},"announcement_url":{"url":"https://api-docs.deepseek.com/news/news260424","retrieved_at":"2026-05-01T07:00:00Z","title":"DeepSeek V4 Preview Release","excerpt":"DeepSeek-V4-Flash: 284B total / 13B active params. Fast, efficient, and economical choice. 1M context length."}}},{"entity_id":"deepseek-v4-pro","fields":{"name":"DeepSeek-V4-Pro","vendor":"DeepSeek","released_at":"2026-04-24","modality":"text, code","access":"open-weights","parameters":"1.6T total / 49B active","context_window":1000000,"announcement_url":"https://api-docs.deepseek.com/news/news260424","benchmark_score":80.6,"benchmark_name":"SWE-Bench Verified"},"evidence":{"name":{"url":"https://api-docs.deepseek.com/news/news260424","retrieved_at":"2026-05-01T07:00:00Z","title":"DeepSeek V4 Preview Release","excerpt":"DeepSeek-V4-Pro: 1.6T total / 49B active params. Performance rivaling the world's top closed-source models. 1M context length."},"vendor":{"url":"https://api-docs.deepseek.com/news/news260424","retrieved_at":"2026-05-01T07:00:00Z","title":"DeepSeek V4 Preview Release","excerpt":"DeepSeek-V4-Pro: 1.6T total / 49B active params. Performance rivaling the world's top closed-source models. 1M context length."},"released_at":{"url":"https://api-docs.deepseek.com/news/news260424","retrieved_at":"2026-05-01T07:00:00Z","title":"DeepSeek V4 Preview Release","excerpt":"DeepSeek-V4-Pro: 1.6T total / 49B active params. Performance rivaling the world's top closed-source models. 1M context length."},"modality":{"url":"https://api-docs.deepseek.com/news/news260424","retrieved_at":"2026-05-01T07:00:00Z","title":"DeepSeek V4 Preview Release","excerpt":"DeepSeek-V4-Pro: 1.6T total / 49B active params. Performance rivaling the world's top closed-source models. 1M context length."},"access":{"url":"https://api-docs.deepseek.com/news/news260424","retrieved_at":"2026-05-01T07:00:00Z","title":"DeepSeek V4 Preview Release","excerpt":"DeepSeek-V4-Pro: 1.6T total / 49B active params. Performance rivaling the world's top closed-source models. 1M context length."},"parameters":{"url":"https://api-docs.deepseek.com/news/news260424","retrieved_at":"2026-05-01T07:00:00Z","title":"DeepSeek V4 Preview Release","excerpt":"DeepSeek-V4-Pro: 1.6T total / 49B active params. Performance rivaling the world's top closed-source models. 1M context length."},"context_window":{"url":"https://api-docs.deepseek.com/news/news260424","retrieved_at":"2026-05-01T07:00:00Z","title":"DeepSeek V4 Preview Release","excerpt":"DeepSeek-V4-Pro: 1.6T total / 49B active params. Performance rivaling the world's top closed-source models. 1M context length."},"announcement_url":{"url":"https://api-docs.deepseek.com/news/news260424","retrieved_at":"2026-05-01T07:00:00Z","title":"DeepSeek V4 Preview Release","excerpt":"DeepSeek-V4-Pro: 1.6T total / 49B active params. Performance rivaling the world's top closed-source models. 1M context length."},"benchmark_score":{"url":"https://deepseekai.guide/models/deepseek-v4/","retrieved_at":"2026-05-02T06:50:40Z","title":"DeepSeek V4: 1M-Token MoE Pro & Flash Tiers Reviewed","excerpt":"SWE-Bench Verified: V4-Pro (reasoning_effort=max): 80.6% | LiveCodeBench: 93.5% | Terminal-Bench 2.0: 67.9% (all DeepSeek-reported from the V4 announcement and technical report)"},"benchmark_name":{"url":"https://deepseekai.guide/models/deepseek-v4/","retrieved_at":"2026-05-02T06:50:40Z","title":"DeepSeek V4: 1M-Token MoE Pro & Flash Tiers Reviewed","excerpt":"SWE-Bench Verified: V4-Pro (reasoning_effort=max): 80.6% (DeepSeek-reported from V4 technical report)"}}},{"entity_id":"google-gemini-3-1-flash-image","fields":{"name":"Gemini 3.1 Flash Image (Nano Banana 2)","vendor":"Google DeepMind","released_at":"2026-04-22","modality":"image","access":"closed","announcement_url":"https://deepmind.google/models/gemini-image/flash/"},"evidence":{"name":{"url":"https://cloud.google.com/blog/products/ai-machine-learning/introducing-gemini-enterprise-agent-platform","retrieved_at":"2026-05-01T07:00:00Z","title":"Introducing Gemini Enterprise Agent Platform | Google Cloud Blog","excerpt":"Gemini 3.1 Flash Image (also known as Nano Banana 2) for creating stunning visual assets, available in Gemini Enterprise Agent Platform."},"vendor":{"url":"https://cloud.google.com/blog/products/ai-machine-learning/introducing-gemini-enterprise-agent-platform","retrieved_at":"2026-05-01T07:00:00Z","title":"Introducing Gemini Enterprise Agent Platform | Google Cloud Blog","excerpt":"Gemini 3.1 Flash Image (also known as Nano Banana 2) for creating stunning visual assets, available in Gemini Enterprise Agent Platform."},"released_at":{"url":"https://cloud.google.com/blog/products/ai-machine-learning/introducing-gemini-enterprise-agent-platform","retrieved_at":"2026-05-01T07:00:00Z","title":"Introducing Gemini Enterprise Agent Platform | Google Cloud Blog","excerpt":"Gemini 3.1 Flash Image (also known as Nano Banana 2) for creating stunning visual assets, available in Gemini Enterprise Agent Platform."},"modality":{"url":"https://cloud.google.com/blog/products/ai-machine-learning/introducing-gemini-enterprise-agent-platform","retrieved_at":"2026-05-01T07:00:00Z","title":"Introducing Gemini Enterprise Agent Platform | Google Cloud Blog","excerpt":"Gemini 3.1 Flash Image (also known as Nano Banana 2) for creating stunning visual assets, available in Gemini Enterprise Agent Platform."},"access":{"url":"https://cloud.google.com/blog/products/ai-machine-learning/introducing-gemini-enterprise-agent-platform","retrieved_at":"2026-05-01T07:00:00Z","title":"Introducing Gemini Enterprise Agent Platform | Google Cloud Blog","excerpt":"Gemini 3.1 Flash Image (also known as Nano Banana 2) for creating stunning visual assets, available in Gemini Enterprise Agent Platform."},"announcement_url":{"url":"https://cloud.google.com/blog/products/ai-machine-learning/introducing-gemini-enterprise-agent-platform","retrieved_at":"2026-05-01T07:00:00Z","title":"Introducing Gemini Enterprise Agent Platform | Google Cloud Blog","excerpt":"Gemini 3.1 Flash Image (also known as Nano Banana 2) for creating stunning visual assets, available in Gemini Enterprise Agent Platform."}}},{"entity_id":"google-gemini-3-1-pro","fields":{"name":"Gemini 3.1 Pro","vendor":"Google DeepMind","released_at":"2026-04-22","modality":"text, image, video, audio, code","access":"closed","context_window":1000000,"benchmark_score":68.5,"benchmark_name":"Terminal-Bench 2.0","announcement_url":"https://deepmind.google/models/gemini/pro/"},"evidence":{"name":{"url":"https://deepmind.google/models/gemini/pro/","retrieved_at":"2026-05-01T07:00:00Z","title":"Gemini 3.1 Pro — Google DeepMind","excerpt":"Gemini 3.1 Pro — Best for complex tasks. Input: Text, Image, Video, Audio, PDF. Output: Text. Input tokens: 1M. Status: Preview."},"vendor":{"url":"https://deepmind.google/models/gemini/pro/","retrieved_at":"2026-05-01T07:00:00Z","title":"Gemini 3.1 Pro — Google DeepMind","excerpt":"Gemini 3.1 Pro — Best for complex tasks. Input: Text, Image, Video, Audio, PDF. Output: Text. Input tokens: 1M. Status: Preview."},"released_at":{"url":"https://deepmind.google/models/gemini/pro/","retrieved_at":"2026-05-01T07:00:00Z","title":"Gemini 3.1 Pro — Google DeepMind","excerpt":"Gemini 3.1 Pro — Best for complex tasks. Input: Text, Image, Video, Audio, PDF. Output: Text. Input tokens: 1M. Status: Preview."},"modality":{"url":"https://deepmind.google/models/gemini/pro/","retrieved_at":"2026-05-01T07:00:00Z","title":"Gemini 3.1 Pro — Google DeepMind","excerpt":"Gemini 3.1 Pro — Best for complex tasks. Input: Text, Image, Video, Audio, PDF. Output: Text. Input tokens: 1M. Status: Preview."},"access":{"url":"https://deepmind.google/models/gemini/pro/","retrieved_at":"2026-05-01T07:00:00Z","title":"Gemini 3.1 Pro — Google DeepMind","excerpt":"Gemini 3.1 Pro — Best for complex tasks. Input: Text, Image, Video, Audio, PDF. Output: Text. Input tokens: 1M. Status: Preview."},"context_window":{"url":"https://deepmind.google/models/gemini/pro/","retrieved_at":"2026-05-01T07:00:00Z","title":"Gemini 3.1 Pro — Google DeepMind","excerpt":"Gemini 3.1 Pro — Best for complex tasks. Input: Text, Image, Video, Audio, PDF. Output: Text. Input tokens: 1M. Status: Preview."},"benchmark_score":{"url":"https://deepmind.google/models/gemini/pro/","retrieved_at":"2026-05-01T07:00:00Z","title":"Gemini 3.1 Pro — Google DeepMind","excerpt":"Gemini 3.1 Pro — Best for complex tasks. Input: Text, Image, Video, Audio, PDF. Output: Text. Input tokens: 1M. Status: Preview."},"benchmark_name":{"url":"https://deepmind.google/models/gemini/pro/","retrieved_at":"2026-05-01T07:00:00Z","title":"Gemini 3.1 Pro — Google DeepMind","excerpt":"Gemini 3.1 Pro — Best for complex tasks. Input: Text, Image, Video, Audio, PDF. Output: Text. Input tokens: 1M. Status: Preview."},"announcement_url":{"url":"https://deepmind.google/models/gemini/pro/","retrieved_at":"2026-05-01T07:00:00Z","title":"Gemini 3.1 Pro — Google DeepMind","excerpt":"Gemini 3.1 Pro — Best for complex tasks. Input: Text, Image, Video, Audio, PDF. Output: Text. Input tokens: 1M. Status: Preview."}}},{"entity_id":"google-gemma-4-26b-a4b","fields":{"name":"Gemma 4 26B-A4B IT (MoE)","vendor":"Google DeepMind","released_at":"2026-04-22","modality":"text, image, audio, code","access":"open-weights","parameters":"26B total / 4B active","benchmark_score":77.1,"benchmark_name":"LiveCodeBench v6","announcement_url":"https://deepmind.google/models/gemma/gemma-4/"},"evidence":{"name":{"url":"https://deepmind.google/models/gemma/gemma-4/","retrieved_at":"2026-05-01T07:00:00Z","title":"Gemma 4 — Google DeepMind","excerpt":"Gemma 4 26B A4B IT Thinking (MoE): AIME 2026 88.3%, LiveCodeBench v6 77.1%, GPQA Diamond 82.3%."},"vendor":{"url":"https://deepmind.google/models/gemma/gemma-4/","retrieved_at":"2026-05-01T07:00:00Z","title":"Gemma 4 — Google DeepMind","excerpt":"Gemma 4 26B A4B IT Thinking (MoE): AIME 2026 88.3%, LiveCodeBench v6 77.1%, GPQA Diamond 82.3%."},"released_at":{"url":"https://deepmind.google/models/gemma/gemma-4/","retrieved_at":"2026-05-01T07:00:00Z","title":"Gemma 4 — Google DeepMind","excerpt":"Gemma 4 26B A4B IT Thinking (MoE): AIME 2026 88.3%, LiveCodeBench v6 77.1%, GPQA Diamond 82.3%."},"modality":{"url":"https://deepmind.google/models/gemma/gemma-4/","retrieved_at":"2026-05-01T07:00:00Z","title":"Gemma 4 — Google DeepMind","excerpt":"Gemma 4 26B A4B IT Thinking (MoE): AIME 2026 88.3%, LiveCodeBench v6 77.1%, GPQA Diamond 82.3%."},"access":{"url":"https://deepmind.google/models/gemma/gemma-4/","retrieved_at":"2026-05-01T07:00:00Z","title":"Gemma 4 — Google DeepMind","excerpt":"Gemma 4 26B A4B IT Thinking (MoE): AIME 2026 88.3%, LiveCodeBench v6 77.1%, GPQA Diamond 82.3%."},"parameters":{"url":"https://deepmind.google/models/gemma/gemma-4/","retrieved_at":"2026-05-01T07:00:00Z","title":"Gemma 4 — Google DeepMind","excerpt":"Gemma 4 26B A4B IT Thinking (MoE): AIME 2026 88.3%, LiveCodeBench v6 77.1%, GPQA Diamond 82.3%."},"benchmark_score":{"url":"https://deepmind.google/models/gemma/gemma-4/","retrieved_at":"2026-05-01T07:00:00Z","title":"Gemma 4 — Google DeepMind","excerpt":"Gemma 4 26B A4B IT Thinking (MoE): AIME 2026 88.3%, LiveCodeBench v6 77.1%, GPQA Diamond 82.3%."},"benchmark_name":{"url":"https://deepmind.google/models/gemma/gemma-4/","retrieved_at":"2026-05-01T07:00:00Z","title":"Gemma 4 — Google DeepMind","excerpt":"Gemma 4 26B A4B IT Thinking (MoE): AIME 2026 88.3%, LiveCodeBench v6 77.1%, GPQA Diamond 82.3%."},"announcement_url":{"url":"https://deepmind.google/models/gemma/gemma-4/","retrieved_at":"2026-05-01T07:00:00Z","title":"Gemma 4 — Google DeepMind","excerpt":"Gemma 4 26B A4B IT Thinking (MoE): AIME 2026 88.3%, LiveCodeBench v6 77.1%, GPQA Diamond 82.3%."}}},{"entity_id":"google-gemma-4-31b","fields":{"name":"Gemma 4 31B IT","vendor":"Google DeepMind","released_at":"2026-04-22","modality":"text, image, audio, code","access":"open-weights","parameters":"31B","benchmark_score":80,"benchmark_name":"LiveCodeBench v6","announcement_url":"https://deepmind.google/models/gemma/gemma-4/"},"evidence":{"name":{"url":"https://deepmind.google/models/gemma/gemma-4/","retrieved_at":"2026-05-01T07:00:00Z","title":"Gemma 4 — Google DeepMind","excerpt":"Gemma 4 31B IT Thinking: AIME 2026 89.2%, LiveCodeBench v6 80.0%, GPQA Diamond 84.3%. Frontier intelligence on personal computers."},"vendor":{"url":"https://deepmind.google/models/gemma/gemma-4/","retrieved_at":"2026-05-01T07:00:00Z","title":"Gemma 4 — Google DeepMind","excerpt":"Gemma 4 31B IT Thinking: AIME 2026 89.2%, LiveCodeBench v6 80.0%, GPQA Diamond 84.3%. Frontier intelligence on personal computers."},"released_at":{"url":"https://deepmind.google/models/gemma/gemma-4/","retrieved_at":"2026-05-01T07:00:00Z","title":"Gemma 4 — Google DeepMind","excerpt":"Gemma 4 31B IT Thinking: AIME 2026 89.2%, LiveCodeBench v6 80.0%, GPQA Diamond 84.3%. Frontier intelligence on personal computers."},"modality":{"url":"https://deepmind.google/models/gemma/gemma-4/","retrieved_at":"2026-05-01T07:00:00Z","title":"Gemma 4 — Google DeepMind","excerpt":"Gemma 4 31B IT Thinking: AIME 2026 89.2%, LiveCodeBench v6 80.0%, GPQA Diamond 84.3%. Frontier intelligence on personal computers."},"access":{"url":"https://deepmind.google/models/gemma/gemma-4/","retrieved_at":"2026-05-01T07:00:00Z","title":"Gemma 4 — Google DeepMind","excerpt":"Gemma 4 31B IT Thinking: AIME 2026 89.2%, LiveCodeBench v6 80.0%, GPQA Diamond 84.3%. Frontier intelligence on personal computers."},"parameters":{"url":"https://deepmind.google/models/gemma/gemma-4/","retrieved_at":"2026-05-01T07:00:00Z","title":"Gemma 4 — Google DeepMind","excerpt":"Gemma 4 31B IT Thinking: AIME 2026 89.2%, LiveCodeBench v6 80.0%, GPQA Diamond 84.3%. Frontier intelligence on personal computers."},"benchmark_score":{"url":"https://deepmind.google/models/gemma/gemma-4/","retrieved_at":"2026-05-01T07:00:00Z","title":"Gemma 4 — Google DeepMind","excerpt":"Gemma 4 31B IT Thinking: AIME 2026 89.2%, LiveCodeBench v6 80.0%, GPQA Diamond 84.3%. Frontier intelligence on personal computers."},"benchmark_name":{"url":"https://deepmind.google/models/gemma/gemma-4/","retrieved_at":"2026-05-01T07:00:00Z","title":"Gemma 4 — Google DeepMind","excerpt":"Gemma 4 31B IT Thinking: AIME 2026 89.2%, LiveCodeBench v6 80.0%, GPQA Diamond 84.3%. Frontier intelligence on personal computers."},"announcement_url":{"url":"https://deepmind.google/models/gemma/gemma-4/","retrieved_at":"2026-05-01T07:00:00Z","title":"Gemma 4 — Google DeepMind","excerpt":"Gemma 4 31B IT Thinking: AIME 2026 89.2%, LiveCodeBench v6 80.0%, GPQA Diamond 84.3%. Frontier intelligence on personal computers."}}},{"entity_id":"google-lyria-3","fields":{"name":"Lyria 3","vendor":"Google DeepMind","released_at":"2026-04-22","modality":"audio","access":"closed","announcement_url":"https://deepmind.google/models/lyria/"},"evidence":{"name":{"url":"https://deepmind.google/models/lyria/","retrieved_at":"2026-05-01T07:00:00Z","title":"Lyria 3 — Google DeepMind","excerpt":"Lyria 3 — Our most advanced music generation model yet, now up to 3 minutes long. Available in Gemini app, AI Studio, Vertex AI."},"vendor":{"url":"https://deepmind.google/models/lyria/","retrieved_at":"2026-05-01T07:00:00Z","title":"Lyria 3 — Google DeepMind","excerpt":"Lyria 3 — Our most advanced music generation model yet, now up to 3 minutes long. Available in Gemini app, AI Studio, Vertex AI."},"released_at":{"url":"https://deepmind.google/models/lyria/","retrieved_at":"2026-05-01T07:00:00Z","title":"Lyria 3 — Google DeepMind","excerpt":"Lyria 3 — Our most advanced music generation model yet, now up to 3 minutes long. Available in Gemini app, AI Studio, Vertex AI."},"modality":{"url":"https://deepmind.google/models/lyria/","retrieved_at":"2026-05-01T07:00:00Z","title":"Lyria 3 — Google DeepMind","excerpt":"Lyria 3 — Our most advanced music generation model yet, now up to 3 minutes long. Available in Gemini app, AI Studio, Vertex AI."},"access":{"url":"https://deepmind.google/models/lyria/","retrieved_at":"2026-05-01T07:00:00Z","title":"Lyria 3 — Google DeepMind","excerpt":"Lyria 3 — Our most advanced music generation model yet, now up to 3 minutes long. Available in Gemini app, AI Studio, Vertex AI."},"announcement_url":{"url":"https://deepmind.google/models/lyria/","retrieved_at":"2026-05-01T07:00:00Z","title":"Lyria 3 — Google DeepMind","excerpt":"Lyria 3 — Our most advanced music generation model yet, now up to 3 minutes long. Available in Gemini app, AI Studio, Vertex AI."}}},{"entity_id":"ibm-granite-4-1-30b","fields":{"name":"Granite 4.1 30B Instruct","vendor":"IBM","released_at":"2026-04-29","modality":"text","access":"open-weights","parameters":"30B","context_window":524288,"announcement_url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models"},"evidence":{"name":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite 4.1 collection: dense decoder-only language models in 3B, 8B, and 30B parameter sizes. Context length up to 512K tokens. Apache 2.0."},"vendor":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite 4.1 collection: dense decoder-only language models in 3B, 8B, and 30B parameter sizes. Context length up to 512K tokens. Apache 2.0."},"released_at":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite 4.1 collection: dense decoder-only language models in 3B, 8B, and 30B parameter sizes. Context length up to 512K tokens. Apache 2.0."},"modality":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite 4.1 collection: dense decoder-only language models in 3B, 8B, and 30B parameter sizes. Context length up to 512K tokens. Apache 2.0."},"access":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite 4.1 collection: dense decoder-only language models in 3B, 8B, and 30B parameter sizes. Context length up to 512K tokens. Apache 2.0."},"parameters":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite 4.1 collection: dense decoder-only language models in 3B, 8B, and 30B parameter sizes. Context length up to 512K tokens. Apache 2.0."},"context_window":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite 4.1 collection: dense decoder-only language models in 3B, 8B, and 30B parameter sizes. Context length up to 512K tokens. Apache 2.0."},"announcement_url":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite 4.1 collection: dense decoder-only language models in 3B, 8B, and 30B parameter sizes. Context length up to 512K tokens. Apache 2.0."}}},{"entity_id":"ibm-granite-4-1-3b","fields":{"name":"Granite 4.1 3B Instruct","vendor":"IBM","released_at":"2026-04-29","modality":"text","access":"open-weights","parameters":"3B","context_window":524288,"announcement_url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models"},"evidence":{"name":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite 4.1 collection: dense decoder-only language models in 3B, 8B, and 30B parameter sizes. Context length up to 512K tokens. Apache 2.0."},"vendor":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite 4.1 collection: dense decoder-only language models in 3B, 8B, and 30B parameter sizes. Context length up to 512K tokens. Apache 2.0."},"released_at":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite 4.1 collection: dense decoder-only language models in 3B, 8B, and 30B parameter sizes. Context length up to 512K tokens. Apache 2.0."},"modality":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite 4.1 collection: dense decoder-only language models in 3B, 8B, and 30B parameter sizes. Context length up to 512K tokens. Apache 2.0."},"access":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite 4.1 collection: dense decoder-only language models in 3B, 8B, and 30B parameter sizes. Context length up to 512K tokens. Apache 2.0."},"parameters":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite 4.1 collection: dense decoder-only language models in 3B, 8B, and 30B parameter sizes. Context length up to 512K tokens. Apache 2.0."},"context_window":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite 4.1 collection: dense decoder-only language models in 3B, 8B, and 30B parameter sizes. Context length up to 512K tokens. Apache 2.0."},"announcement_url":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite 4.1 collection: dense decoder-only language models in 3B, 8B, and 30B parameter sizes. Context length up to 512K tokens. Apache 2.0."}}},{"entity_id":"ibm-granite-4-1-8b","fields":{"name":"Granite 4.1 8B Instruct","vendor":"IBM","released_at":"2026-04-29","modality":"text","access":"open-weights","parameters":"8B","context_window":524288,"announcement_url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models"},"evidence":{"name":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite 4.1 8B instruct model consistently matches or outperforms the Granite 4.0 32B MoE. Context length up to 512K tokens. Apache 2.0 license."},"vendor":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite 4.1 8B instruct model consistently matches or outperforms the Granite 4.0 32B MoE. Context length up to 512K tokens. Apache 2.0 license."},"released_at":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite 4.1 8B instruct model consistently matches or outperforms the Granite 4.0 32B MoE. Context length up to 512K tokens. Apache 2.0 license."},"modality":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite 4.1 8B instruct model consistently matches or outperforms the Granite 4.0 32B MoE. Context length up to 512K tokens. Apache 2.0 license."},"access":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite 4.1 8B instruct model consistently matches or outperforms the Granite 4.0 32B MoE. Context length up to 512K tokens. Apache 2.0 license."},"parameters":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite 4.1 8B instruct model consistently matches or outperforms the Granite 4.0 32B MoE. Context length up to 512K tokens. Apache 2.0 license."},"context_window":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite 4.1 8B instruct model consistently matches or outperforms the Granite 4.0 32B MoE. Context length up to 512K tokens. Apache 2.0 license."},"announcement_url":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite 4.1 8B instruct model consistently matches or outperforms the Granite 4.0 32B MoE. Context length up to 512K tokens. Apache 2.0 license."}}},{"entity_id":"ibm-granite-4-1-speech-2b","fields":{"name":"Granite Speech 4.1 2B","vendor":"IBM","released_at":"2026-04-29","modality":"audio, text","access":"open-weights","parameters":"2B","benchmark_score":5.33,"benchmark_name":"OpenASR WER (%)","announcement_url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models"},"evidence":{"name":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite Speech 4.1 2B achieves a 5.33% word-error rate (WER), placing it among the top models on the OpenASR Leaderboard. Apache 2.0."},"vendor":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite Speech 4.1 2B achieves a 5.33% word-error rate (WER), placing it among the top models on the OpenASR Leaderboard. Apache 2.0."},"released_at":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite Speech 4.1 2B achieves a 5.33% word-error rate (WER), placing it among the top models on the OpenASR Leaderboard. Apache 2.0."},"modality":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite Speech 4.1 2B achieves a 5.33% word-error rate (WER), placing it among the top models on the OpenASR Leaderboard. Apache 2.0."},"access":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite Speech 4.1 2B achieves a 5.33% word-error rate (WER), placing it among the top models on the OpenASR Leaderboard. Apache 2.0."},"parameters":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite Speech 4.1 2B achieves a 5.33% word-error rate (WER), placing it among the top models on the OpenASR Leaderboard. Apache 2.0."},"benchmark_score":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite Speech 4.1 2B achieves a 5.33% word-error rate (WER), placing it among the top models on the OpenASR Leaderboard. Apache 2.0."},"benchmark_name":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite Speech 4.1 2B achieves a 5.33% word-error rate (WER), placing it among the top models on the OpenASR Leaderboard. Apache 2.0."},"announcement_url":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite Speech 4.1 2B achieves a 5.33% word-error rate (WER), placing it among the top models on the OpenASR Leaderboard. Apache 2.0."}}},{"entity_id":"ibm-granite-4-1-vision","fields":{"name":"Granite Vision 4.1","vendor":"IBM","released_at":"2026-04-29","modality":"vision, text","access":"open-weights","announcement_url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","parameters":"4B"},"evidence":{"name":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite Vision 4.1: a vision-language model designed for document understanding tasks — tables, charts, key-value pair extraction. Apache 2.0."},"vendor":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite Vision 4.1: a vision-language model designed for document understanding tasks — tables, charts, key-value pair extraction. Apache 2.0."},"released_at":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite Vision 4.1: a vision-language model designed for document understanding tasks — tables, charts, key-value pair extraction. Apache 2.0."},"modality":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite Vision 4.1: a vision-language model designed for document understanding tasks — tables, charts, key-value pair extraction. Apache 2.0."},"access":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite Vision 4.1: a vision-language model designed for document understanding tasks — tables, charts, key-value pair extraction. Apache 2.0."},"announcement_url":{"url":"https://research.ibm.com/blog/granite-4-1-ai-foundation-models","retrieved_at":"2026-04-30","title":"Introducing the IBM Granite 4.1 family of models - IBM Research","excerpt":"Granite Vision 4.1: a vision-language model designed for document understanding tasks — tables, charts, key-value pair extraction. Apache 2.0."},"parameters":{"url":"https://huggingface.co/ibm-granite/granite-vision-4.1-4b","retrieved_at":"2026-05-03T07:00:00Z","title":"ibm-granite/granite-vision-4.1-4b - Hugging Face","excerpt":"Language model: Granite-4.1 (3B) with LoRA. Model size 4B params (3,997,206,464 parameters). Downloads last month 3,406."}}},{"entity_id":"microsoft-vibevoice-asr","fields":{"name":"VibeVoice-ASR","vendor":"Microsoft","released_at":"2026-01-21","modality":"audio","access":"open-weights","parameters":"7B","context_window":65536,"announcement_url":"https://github.com/microsoft/VibeVoice"},"evidence":{"name":{"url":"https://github.com/microsoft/VibeVoice","retrieved_at":"2026-05-05T06:55:00Z","title":"microsoft/VibeVoice","excerpt":"VibeVoice-ASR: a 7B-parameter speech recognition model that processes up to 60 minutes (64K-token context) of continuous audio in a single inference pass, released January 21 2026. Built on Qwen2.5 base with continuous speech tokenizers at 7.5 Hz frame rate. Integrated into Hugging Face Transformers on March 6, 2026."},"vendor":{"url":"https://github.com/microsoft/VibeVoice","retrieved_at":"2026-05-05T06:55:00Z","title":"microsoft/VibeVoice","excerpt":"VibeVoice-ASR: a 7B-parameter speech recognition model that processes up to 60 minutes (64K-token context) of continuous audio in a single inference pass, released January 21 2026. Built on Qwen2.5 base with continuous speech tokenizers at 7.5 Hz frame rate. Integrated into Hugging Face Transformers on March 6, 2026."},"released_at":{"url":"https://github.com/microsoft/VibeVoice","retrieved_at":"2026-05-05T06:55:00Z","title":"microsoft/VibeVoice","excerpt":"VibeVoice-ASR: a 7B-parameter speech recognition model that processes up to 60 minutes (64K-token context) of continuous audio in a single inference pass, released January 21 2026. Built on Qwen2.5 base with continuous speech tokenizers at 7.5 Hz frame rate. Integrated into Hugging Face Transformers on March 6, 2026."},"modality":{"url":"https://github.com/microsoft/VibeVoice","retrieved_at":"2026-05-05T06:55:00Z","title":"microsoft/VibeVoice","excerpt":"VibeVoice-ASR: a 7B-parameter speech recognition model that processes up to 60 minutes (64K-token context) of continuous audio in a single inference pass, released January 21 2026. Built on Qwen2.5 base with continuous speech tokenizers at 7.5 Hz frame rate. Integrated into Hugging Face Transformers on March 6, 2026."},"access":{"url":"https://github.com/microsoft/VibeVoice","retrieved_at":"2026-05-05T06:55:00Z","title":"microsoft/VibeVoice","excerpt":"VibeVoice-ASR: a 7B-parameter speech recognition model that processes up to 60 minutes (64K-token context) of continuous audio in a single inference pass, released January 21 2026. Built on Qwen2.5 base with continuous speech tokenizers at 7.5 Hz frame rate. Integrated into Hugging Face Transformers on March 6, 2026."},"parameters":{"url":"https://github.com/microsoft/VibeVoice","retrieved_at":"2026-05-05T06:55:00Z","title":"microsoft/VibeVoice","excerpt":"VibeVoice-ASR: a 7B-parameter speech recognition model that processes up to 60 minutes (64K-token context) of continuous audio in a single inference pass, released January 21 2026. Built on Qwen2.5 base with continuous speech tokenizers at 7.5 Hz frame rate. Integrated into Hugging Face Transformers on March 6, 2026."},"context_window":{"url":"https://github.com/microsoft/VibeVoice","retrieved_at":"2026-05-05T06:55:00Z","title":"microsoft/VibeVoice","excerpt":"VibeVoice-ASR: a 7B-parameter speech recognition model that processes up to 60 minutes (64K-token context) of continuous audio in a single inference pass, released January 21 2026. Built on Qwen2.5 base with continuous speech tokenizers at 7.5 Hz frame rate. Integrated into Hugging Face Transformers on March 6, 2026."},"announcement_url":{"url":"https://github.com/microsoft/VibeVoice","retrieved_at":"2026-05-05T06:55:00Z","title":"microsoft/VibeVoice","excerpt":"VibeVoice-ASR: a 7B-parameter speech recognition model that processes up to 60 minutes (64K-token context) of continuous audio in a single inference pass, released January 21 2026. Built on Qwen2.5 base with continuous speech tokenizers at 7.5 Hz frame rate. Integrated into Hugging Face Transformers on March 6, 2026."}}},{"entity_id":"mistral-mistral-medium-3-5","fields":{"name":"Mistral Medium 3.5","vendor":"Mistral AI","released_at":"2026-04-29","modality":"text, code, vision","access":"open-weights","context_window":262144,"parameters":"128B","benchmark_score":77.6,"benchmark_name":"SWE-Bench Verified","announcement_url":"https://mistral.ai/news/vibe-remote-agents-mistral-medium-3-5"},"evidence":{"name":{"url":"https://mistral.ai/news/vibe-remote-agents-mistral-medium-3-5","retrieved_at":"2026-04-30","title":"Remote agents in Vibe. Powered by Mistral Medium 3.5. | Mistral AI","excerpt":"Mistral Medium 3.5: a dense 128B model with a 256k context window. Scores 77.6% on SWE-Bench Verified. Released as open weights, under a modified MIT license."},"vendor":{"url":"https://mistral.ai/news/vibe-remote-agents-mistral-medium-3-5","retrieved_at":"2026-04-30","title":"Remote agents in Vibe. Powered by Mistral Medium 3.5. | Mistral AI","excerpt":"Mistral Medium 3.5: a dense 128B model with a 256k context window. Scores 77.6% on SWE-Bench Verified. Released as open weights, under a modified MIT license."},"released_at":{"url":"https://mistral.ai/news/vibe-remote-agents-mistral-medium-3-5","retrieved_at":"2026-04-30","title":"Remote agents in Vibe. Powered by Mistral Medium 3.5. | Mistral AI","excerpt":"Mistral Medium 3.5: a dense 128B model with a 256k context window. Scores 77.6% on SWE-Bench Verified. Released as open weights, under a modified MIT license."},"modality":{"url":"https://mistral.ai/news/vibe-remote-agents-mistral-medium-3-5","retrieved_at":"2026-04-30","title":"Remote agents in Vibe. Powered by Mistral Medium 3.5. | Mistral AI","excerpt":"Mistral Medium 3.5: a dense 128B model with a 256k context window. Scores 77.6% on SWE-Bench Verified. Released as open weights, under a modified MIT license."},"access":{"url":"https://mistral.ai/news/vibe-remote-agents-mistral-medium-3-5","retrieved_at":"2026-04-30","title":"Remote agents in Vibe. Powered by Mistral Medium 3.5. | Mistral AI","excerpt":"Mistral Medium 3.5: a dense 128B model with a 256k context window. Scores 77.6% on SWE-Bench Verified. Released as open weights, under a modified MIT license."},"context_window":{"url":"https://mistral.ai/news/vibe-remote-agents-mistral-medium-3-5","retrieved_at":"2026-04-30","title":"Remote agents in Vibe. Powered by Mistral Medium 3.5. | Mistral AI","excerpt":"Mistral Medium 3.5: a dense 128B model with a 256k context window. Scores 77.6% on SWE-Bench Verified. Released as open weights, under a modified MIT license."},"parameters":{"url":"https://mistral.ai/news/vibe-remote-agents-mistral-medium-3-5","retrieved_at":"2026-04-30","title":"Remote agents in Vibe. Powered by Mistral Medium 3.5. | Mistral AI","excerpt":"Mistral Medium 3.5: a dense 128B model with a 256k context window. Scores 77.6% on SWE-Bench Verified. Released as open weights, under a modified MIT license."},"benchmark_score":{"url":"https://mistral.ai/news/vibe-remote-agents-mistral-medium-3-5","retrieved_at":"2026-04-30","title":"Remote agents in Vibe. Powered by Mistral Medium 3.5. | Mistral AI","excerpt":"Mistral Medium 3.5: a dense 128B model with a 256k context window. Scores 77.6% on SWE-Bench Verified. Released as open weights, under a modified MIT license."},"benchmark_name":{"url":"https://mistral.ai/news/vibe-remote-agents-mistral-medium-3-5","retrieved_at":"2026-04-30","title":"Remote agents in Vibe. Powered by Mistral Medium 3.5. | Mistral AI","excerpt":"Mistral Medium 3.5: a dense 128B model with a 256k context window. Scores 77.6% on SWE-Bench Verified. Released as open weights, under a modified MIT license."},"announcement_url":{"url":"https://mistral.ai/news/vibe-remote-agents-mistral-medium-3-5","retrieved_at":"2026-04-30","title":"Remote agents in Vibe. Powered by Mistral Medium 3.5. | Mistral AI","excerpt":"Mistral Medium 3.5: a dense 128B model with a 256k context window. Scores 77.6% on SWE-Bench Verified. Released as open weights, under a modified MIT license."}}},{"entity_id":"moonshot-kimi-k2","fields":{"name":"Kimi K2","vendor":"Moonshot AI","released_at":"2026-01-20","modality":"text, image, code","access":"open-weights","parameters":"1.04T total / 32B active","context_window":2000000,"benchmark_score":65.8,"benchmark_name":"SWE-Bench Verified","announcement_url":"https://huggingface.co/moonshotai/Kimi-K2"},"evidence":{"name":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"Kimi K2 released 2026-01-20. Open-weight frontier MoE, 1.04T params (32B active), 2M token context, 65.8% SWE-bench Verified. First open-weight model #1 on LMSYS Arena."},"vendor":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"Kimi K2 released 2026-01-20. Open-weight frontier MoE, 1.04T params (32B active), 2M token context, 65.8% SWE-bench Verified. First open-weight model #1 on LMSYS Arena."},"released_at":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"Kimi K2 released 2026-01-20. Open-weight frontier MoE, 1.04T params (32B active), 2M token context, 65.8% SWE-bench Verified. First open-weight model #1 on LMSYS Arena."},"modality":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"Kimi K2 released 2026-01-20. Open-weight frontier MoE, 1.04T params (32B active), 2M token context, 65.8% SWE-bench Verified. First open-weight model #1 on LMSYS Arena."},"access":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"Kimi K2 released 2026-01-20. Open-weight frontier MoE, 1.04T params (32B active), 2M token context, 65.8% SWE-bench Verified. First open-weight model #1 on LMSYS Arena."},"parameters":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"Kimi K2 released 2026-01-20. Open-weight frontier MoE, 1.04T params (32B active), 2M token context, 65.8% SWE-bench Verified. First open-weight model #1 on LMSYS Arena."},"context_window":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"Kimi K2 released 2026-01-20. Open-weight frontier MoE, 1.04T params (32B active), 2M token context, 65.8% SWE-bench Verified. First open-weight model #1 on LMSYS Arena."},"benchmark_score":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"Kimi K2 released 2026-01-20. Open-weight frontier MoE, 1.04T params (32B active), 2M token context, 65.8% SWE-bench Verified. First open-weight model #1 on LMSYS Arena."},"benchmark_name":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"Kimi K2 released 2026-01-20. Open-weight frontier MoE, 1.04T params (32B active), 2M token context, 65.8% SWE-bench Verified. First open-weight model #1 on LMSYS Arena."},"announcement_url":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"Kimi K2 released 2026-01-20. Open-weight frontier MoE, 1.04T params (32B active), 2M token context, 65.8% SWE-bench Verified. First open-weight model #1 on LMSYS Arena."}}},{"entity_id":"moonshot-kimi-k2-6","fields":{"name":"Kimi K2.6","vendor":"Moonshot AI","released_at":"2026-04-20","modality":"text, code","access":"open-weights","benchmark_score":80.2,"benchmark_name":"SWE-Bench Verified","announcement_url":"https://huggingface.co/moonshotai/Kimi-K2.6","context_window":262144,"parameters":"1T total / 32B active"},"evidence":{"name":{"url":"https://www.buildfastwithai.com/blogs/best-ai-models-may-2026-leaderboard","retrieved_at":"2026-05-03T07:00:00Z","title":"Best AI Models: April + May 2026 Leaderboard (GPT-5.5, Claude Opus 4.7, DeepSeek V4)","excerpt":"Kimi K2.6 beat GPT-5.4 on SWE-Bench Pro at $0.60 per million output tokens and achieved Tier A (87/100) on real-world coding benchmarks — the only Chinese model to reach that tier. Supports 300-agent parallel swarm orchestration."},"vendor":{"url":"https://www.buildfastwithai.com/blogs/best-ai-models-may-2026-leaderboard","retrieved_at":"2026-05-03T07:00:00Z","title":"Best AI Models: April + May 2026 Leaderboard (GPT-5.5, Claude Opus 4.7, DeepSeek V4)","excerpt":"Kimi K2.6 beat GPT-5.4 on SWE-Bench Pro at $0.60 per million output tokens and achieved Tier A (87/100) on real-world coding benchmarks — the only Chinese model to reach that tier. Supports 300-agent parallel swarm orchestration."},"released_at":{"url":"https://www.buildfastwithai.com/blogs/best-ai-models-may-2026-leaderboard","retrieved_at":"2026-05-03T07:00:00Z","title":"Best AI Models: April + May 2026 Leaderboard (GPT-5.5, Claude Opus 4.7, DeepSeek V4)","excerpt":"Kimi K2.6 beat GPT-5.4 on SWE-Bench Pro at $0.60 per million output tokens and achieved Tier A (87/100) on real-world coding benchmarks — the only Chinese model to reach that tier. Supports 300-agent parallel swarm orchestration."},"modality":{"url":"https://www.buildfastwithai.com/blogs/best-ai-models-may-2026-leaderboard","retrieved_at":"2026-05-03T07:00:00Z","title":"Best AI Models: April + May 2026 Leaderboard (GPT-5.5, Claude Opus 4.7, DeepSeek V4)","excerpt":"Kimi K2.6 beat GPT-5.4 on SWE-Bench Pro at $0.60 per million output tokens and achieved Tier A (87/100) on real-world coding benchmarks — the only Chinese model to reach that tier. Supports 300-agent parallel swarm orchestration."},"access":{"url":"https://awesomeagents.ai/news/kimi-k2-6-agent-swarm-open-weight/","retrieved_at":"2026-05-05T06:56:00Z","title":"Kimi K2.6 - Open Weights, 300 Agents, Top Coding Score","excerpt":"Moonshot AI released Kimi K2.6 on April 20 with open weights on HuggingFace under a Modified MIT license (moonshotai/Kimi-K2.6). 1T total / 32B active MoE. SWE-Bench Verified 80.2%, SWE-Bench Pro 58.6%. Context window 256K."},"benchmark_score":{"url":"https://awesomeagents.ai/news/kimi-k2-6-agent-swarm-open-weight/","retrieved_at":"2026-05-05T06:56:00Z","title":"Kimi K2.6 - Open Weights, 300 Agents, Top Coding Score","excerpt":"Moonshot AI released Kimi K2.6 on April 20 with open weights on HuggingFace under a Modified MIT license (moonshotai/Kimi-K2.6). 1T total / 32B active MoE. SWE-Bench Verified 80.2%, SWE-Bench Pro 58.6%. Context window 256K."},"benchmark_name":{"url":"https://awesomeagents.ai/news/kimi-k2-6-agent-swarm-open-weight/","retrieved_at":"2026-05-05T06:56:00Z","title":"Kimi K2.6 - Open Weights, 300 Agents, Top Coding Score","excerpt":"Moonshot AI released Kimi K2.6 on April 20 with open weights on HuggingFace under a Modified MIT license (moonshotai/Kimi-K2.6). 1T total / 32B active MoE. SWE-Bench Verified 80.2%, SWE-Bench Pro 58.6%. Context window 256K."},"announcement_url":{"url":"https://awesomeagents.ai/news/kimi-k2-6-agent-swarm-open-weight/","retrieved_at":"2026-05-05T06:56:00Z","title":"Kimi K2.6 - Open Weights, 300 Agents, Top Coding Score","excerpt":"Moonshot AI released Kimi K2.6 on April 20 with open weights on HuggingFace under a Modified MIT license (moonshotai/Kimi-K2.6). 1T total / 32B active MoE. SWE-Bench Verified 80.2%, SWE-Bench Pro 58.6%. Context window 256K."},"context_window":{"url":"https://awesomeagents.ai/news/kimi-k2-6-agent-swarm-open-weight/","retrieved_at":"2026-05-05T06:56:00Z","title":"Kimi K2.6 - Open Weights, 300 Agents, Top Coding Score","excerpt":"Moonshot AI released Kimi K2.6 on April 20 with open weights on HuggingFace under a Modified MIT license (moonshotai/Kimi-K2.6). 1T total / 32B active MoE. SWE-Bench Verified 80.2%, SWE-Bench Pro 58.6%. Context window 256K."},"parameters":{"url":"https://awesomeagents.ai/news/kimi-k2-6-agent-swarm-open-weight/","retrieved_at":"2026-05-05T06:56:00Z","title":"Kimi K2.6 - Open Weights, 300 Agents, Top Coding Score","excerpt":"Moonshot AI released Kimi K2.6 on April 20 with open weights on HuggingFace under a Modified MIT license (moonshotai/Kimi-K2.6). 1T total / 32B active MoE. SWE-Bench Verified 80.2%, SWE-Bench Pro 58.6%. Context window 256K."}}},{"entity_id":"nvidia-nemotron-3-nano-omni-30b","fields":{"name":"Nemotron 3 Nano Omni 30B-A3B","vendor":"NVIDIA","released_at":"2026-04-28","modality":"text, vision, audio, video","access":"open-weights","parameters":"30B-A3B","benchmark_score":57.5,"benchmark_name":"MMLongBench-Doc","announcement_url":"https://huggingface.co/blog/nvidia/nemotron-3-nano-omni-multimodal-intelligence"},"evidence":{"name":{"url":"https://huggingface.co/blog/nvidia/nemotron-3-nano-omni-multimodal-intelligence","retrieved_at":"2026-04-30","title":"Introducing NVIDIA Nemotron 3 Nano Omni: Long-Context Multimodal Intelligence | Hugging Face","excerpt":"NVIDIA Nemotron 3 Nano Omni is a new omni-modal understanding model. 30B-A3B Mamba-Transformer MoE. MMLongBench-Doc: 57.5, VoiceBench: 89.4. Published April 28, 2026."},"vendor":{"url":"https://huggingface.co/blog/nvidia/nemotron-3-nano-omni-multimodal-intelligence","retrieved_at":"2026-04-30","title":"Introducing NVIDIA Nemotron 3 Nano Omni: Long-Context Multimodal Intelligence | Hugging Face","excerpt":"NVIDIA Nemotron 3 Nano Omni is a new omni-modal understanding model. 30B-A3B Mamba-Transformer MoE. MMLongBench-Doc: 57.5, VoiceBench: 89.4. Published April 28, 2026."},"released_at":{"url":"https://huggingface.co/blog/nvidia/nemotron-3-nano-omni-multimodal-intelligence","retrieved_at":"2026-04-30","title":"Introducing NVIDIA Nemotron 3 Nano Omni: Long-Context Multimodal Intelligence | Hugging Face","excerpt":"NVIDIA Nemotron 3 Nano Omni is a new omni-modal understanding model. 30B-A3B Mamba-Transformer MoE. MMLongBench-Doc: 57.5, VoiceBench: 89.4. Published April 28, 2026."},"modality":{"url":"https://huggingface.co/blog/nvidia/nemotron-3-nano-omni-multimodal-intelligence","retrieved_at":"2026-04-30","title":"Introducing NVIDIA Nemotron 3 Nano Omni: Long-Context Multimodal Intelligence | Hugging Face","excerpt":"NVIDIA Nemotron 3 Nano Omni is a new omni-modal understanding model. 30B-A3B Mamba-Transformer MoE. MMLongBench-Doc: 57.5, VoiceBench: 89.4. Published April 28, 2026."},"access":{"url":"https://huggingface.co/blog/nvidia/nemotron-3-nano-omni-multimodal-intelligence","retrieved_at":"2026-04-30","title":"Introducing NVIDIA Nemotron 3 Nano Omni: Long-Context Multimodal Intelligence | Hugging Face","excerpt":"NVIDIA Nemotron 3 Nano Omni is a new omni-modal understanding model. 30B-A3B Mamba-Transformer MoE. MMLongBench-Doc: 57.5, VoiceBench: 89.4. Published April 28, 2026."},"parameters":{"url":"https://huggingface.co/blog/nvidia/nemotron-3-nano-omni-multimodal-intelligence","retrieved_at":"2026-04-30","title":"Introducing NVIDIA Nemotron 3 Nano Omni: Long-Context Multimodal Intelligence | Hugging Face","excerpt":"NVIDIA Nemotron 3 Nano Omni is a new omni-modal understanding model. 30B-A3B Mamba-Transformer MoE. MMLongBench-Doc: 57.5, VoiceBench: 89.4. Published April 28, 2026."},"benchmark_score":{"url":"https://huggingface.co/blog/nvidia/nemotron-3-nano-omni-multimodal-intelligence","retrieved_at":"2026-04-30","title":"Introducing NVIDIA Nemotron 3 Nano Omni: Long-Context Multimodal Intelligence | Hugging Face","excerpt":"NVIDIA Nemotron 3 Nano Omni is a new omni-modal understanding model. 30B-A3B Mamba-Transformer MoE. MMLongBench-Doc: 57.5, VoiceBench: 89.4. Published April 28, 2026."},"benchmark_name":{"url":"https://huggingface.co/blog/nvidia/nemotron-3-nano-omni-multimodal-intelligence","retrieved_at":"2026-04-30","title":"Introducing NVIDIA Nemotron 3 Nano Omni: Long-Context Multimodal Intelligence | Hugging Face","excerpt":"NVIDIA Nemotron 3 Nano Omni is a new omni-modal understanding model. 30B-A3B Mamba-Transformer MoE. MMLongBench-Doc: 57.5, VoiceBench: 89.4. Published April 28, 2026."},"announcement_url":{"url":"https://huggingface.co/blog/nvidia/nemotron-3-nano-omni-multimodal-intelligence","retrieved_at":"2026-04-30","title":"Introducing NVIDIA Nemotron 3 Nano Omni: Long-Context Multimodal Intelligence | Hugging Face","excerpt":"NVIDIA Nemotron 3 Nano Omni is a new omni-modal understanding model. 30B-A3B Mamba-Transformer MoE. MMLongBench-Doc: 57.5, VoiceBench: 89.4. Published April 28, 2026."}}},{"entity_id":"openai-gpt-5-3-codex","fields":{"name":"GPT-5.3 Codex","vendor":"OpenAI","released_at":"2026-02-05","modality":"text, code","access":"closed","context_window":400000,"benchmark_score":82.4,"benchmark_name":"SWE-Bench Verified","announcement_url":"https://openai.com/index/introducing-gpt-5-3-codex/"},"evidence":{"name":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"GPT-5.3 Codex released 2026-02-05. Coding-specialized variant of GPT-5.3, tuned for agentic IDE workflows. ~200B MoE, 400K context, 82.4% SWE-bench Verified."},"vendor":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"GPT-5.3 Codex released 2026-02-05. Coding-specialized variant of GPT-5.3, tuned for agentic IDE workflows. ~200B MoE, 400K context, 82.4% SWE-bench Verified."},"released_at":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"GPT-5.3 Codex released 2026-02-05. Coding-specialized variant of GPT-5.3, tuned for agentic IDE workflows. ~200B MoE, 400K context, 82.4% SWE-bench Verified."},"modality":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"GPT-5.3 Codex released 2026-02-05. Coding-specialized variant of GPT-5.3, tuned for agentic IDE workflows. ~200B MoE, 400K context, 82.4% SWE-bench Verified."},"access":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"GPT-5.3 Codex released 2026-02-05. Coding-specialized variant of GPT-5.3, tuned for agentic IDE workflows. ~200B MoE, 400K context, 82.4% SWE-bench Verified."},"context_window":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"GPT-5.3 Codex released 2026-02-05. Coding-specialized variant of GPT-5.3, tuned for agentic IDE workflows. ~200B MoE, 400K context, 82.4% SWE-bench Verified."},"benchmark_score":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"GPT-5.3 Codex released 2026-02-05. Coding-specialized variant of GPT-5.3, tuned for agentic IDE workflows. ~200B MoE, 400K context, 82.4% SWE-bench Verified."},"benchmark_name":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"GPT-5.3 Codex released 2026-02-05. Coding-specialized variant of GPT-5.3, tuned for agentic IDE workflows. ~200B MoE, 400K context, 82.4% SWE-bench Verified."},"announcement_url":{"url":"https://aiflashreport.com/model-releases.html","retrieved_at":"2026-05-04T07:10:00Z","title":"AI Model Release Timeline 2025-2026","excerpt":"GPT-5.3 Codex released 2026-02-05. Coding-specialized variant of GPT-5.3, tuned for agentic IDE workflows. ~200B MoE, 400K context, 82.4% SWE-bench Verified."}}},{"entity_id":"openai-gpt-5-3-instant","fields":{"name":"GPT-5.3 Instant","vendor":"OpenAI","released_at":"2026-04-26","modality":"text, code, vision","access":"closed","context_window":131072,"announcement_url":"https://help.openai.com/en/articles/11909943-gpt-53-and-gpt-55-in-chatgpt"},"evidence":{"name":{"url":"https://help.openai.com/en/articles/11909943-gpt-53-and-gpt-55-in-chatgpt","retrieved_at":"2026-05-03T07:00:00Z","title":"GPT-5.3 and GPT-5.5 in ChatGPT | OpenAI Help Center","excerpt":"GPT-5.3 in ChatGPT is the default for all logged-in users. GPT-5.3 Instant is a fast and powerful workhorse for everyday work and learning. Plus/Business context: 32K; Pro/Enterprise: 128K."},"vendor":{"url":"https://help.openai.com/en/articles/11909943-gpt-53-and-gpt-55-in-chatgpt","retrieved_at":"2026-05-03T07:00:00Z","title":"GPT-5.3 and GPT-5.5 in ChatGPT | OpenAI Help Center","excerpt":"GPT-5.3 in ChatGPT is the default for all logged-in users. GPT-5.3 Instant is a fast and powerful workhorse for everyday work and learning. Plus/Business context: 32K; Pro/Enterprise: 128K."},"released_at":{"url":"https://help.openai.com/en/articles/11909943-gpt-53-and-gpt-55-in-chatgpt","retrieved_at":"2026-05-03T07:00:00Z","title":"GPT-5.3 and GPT-5.5 in ChatGPT | OpenAI Help Center","excerpt":"GPT-5.3 in ChatGPT is the default for all logged-in users. GPT-5.3 Instant is a fast and powerful workhorse for everyday work and learning. Plus/Business context: 32K; Pro/Enterprise: 128K."},"modality":{"url":"https://help.openai.com/en/articles/11909943-gpt-53-and-gpt-55-in-chatgpt","retrieved_at":"2026-05-03T07:00:00Z","title":"GPT-5.3 and GPT-5.5 in ChatGPT | OpenAI Help Center","excerpt":"GPT-5.3 in ChatGPT is the default for all logged-in users. GPT-5.3 Instant is a fast and powerful workhorse for everyday work and learning. Plus/Business context: 32K; Pro/Enterprise: 128K."},"access":{"url":"https://help.openai.com/en/articles/11909943-gpt-53-and-gpt-55-in-chatgpt","retrieved_at":"2026-05-03T07:00:00Z","title":"GPT-5.3 and GPT-5.5 in ChatGPT | OpenAI Help Center","excerpt":"GPT-5.3 in ChatGPT is the default for all logged-in users. GPT-5.3 Instant is a fast and powerful workhorse for everyday work and learning. Plus/Business context: 32K; Pro/Enterprise: 128K."},"context_window":{"url":"https://help.openai.com/en/articles/11909943-gpt-53-and-gpt-55-in-chatgpt","retrieved_at":"2026-05-03T07:00:00Z","title":"GPT-5.3 and GPT-5.5 in ChatGPT | OpenAI Help Center","excerpt":"GPT-5.3 in ChatGPT is the default for all logged-in users. GPT-5.3 Instant is a fast and powerful workhorse for everyday work and learning. Plus/Business context: 32K; Pro/Enterprise: 128K."},"announcement_url":{"url":"https://help.openai.com/en/articles/11909943-gpt-53-and-gpt-55-in-chatgpt","retrieved_at":"2026-05-03T07:00:00Z","title":"GPT-5.3 and GPT-5.5 in ChatGPT | OpenAI Help Center","excerpt":"GPT-5.3 in ChatGPT is the default for all logged-in users. GPT-5.3 Instant is a fast and powerful workhorse for everyday work and learning. Plus/Business context: 32K; Pro/Enterprise: 128K."}}},{"entity_id":"openai-gpt-5-5","fields":{"name":"GPT-5.5","vendor":"OpenAI","released_at":"2026-04-23","modality":"text, code, vision","access":"closed","context_window":1000000,"benchmark_score":88.7,"benchmark_name":"SWE-Bench Verified","announcement_url":"https://openai.com/index/introducing-gpt-5-5/"},"evidence":{"name":{"url":"https://openai.com/index/introducing-gpt-5-5/","retrieved_at":"2026-04-30","title":"Introducing GPT-5.5 | OpenAI","excerpt":"We're releasing GPT-5.5, our smartest and most intuitive to use model yet. Terminal-Bench 2.0: 82.7%. Context window: 1M tokens."},"vendor":{"url":"https://openai.com/index/introducing-gpt-5-5/","retrieved_at":"2026-04-30","title":"Introducing GPT-5.5 | OpenAI","excerpt":"We're releasing GPT-5.5, our smartest and most intuitive to use model yet. Terminal-Bench 2.0: 82.7%. Context window: 1M tokens."},"released_at":{"url":"https://openai.com/index/introducing-gpt-5-5/","retrieved_at":"2026-04-30","title":"Introducing GPT-5.5 | OpenAI","excerpt":"We're releasing GPT-5.5, our smartest and most intuitive to use model yet. Terminal-Bench 2.0: 82.7%. Context window: 1M tokens."},"modality":{"url":"https://openai.com/index/introducing-gpt-5-5/","retrieved_at":"2026-04-30","title":"Introducing GPT-5.5 | OpenAI","excerpt":"We're releasing GPT-5.5, our smartest and most intuitive to use model yet. Terminal-Bench 2.0: 82.7%. Context window: 1M tokens."},"access":{"url":"https://openai.com/index/introducing-gpt-5-5/","retrieved_at":"2026-04-30","title":"Introducing GPT-5.5 | OpenAI","excerpt":"We're releasing GPT-5.5, our smartest and most intuitive to use model yet. Terminal-Bench 2.0: 82.7%. Context window: 1M tokens."},"context_window":{"url":"https://openai.com/index/introducing-gpt-5-5/","retrieved_at":"2026-04-30","title":"Introducing GPT-5.5 | OpenAI","excerpt":"We're releasing GPT-5.5, our smartest and most intuitive to use model yet. Terminal-Bench 2.0: 82.7%. Context window: 1M tokens."},"benchmark_score":{"url":"https://www.buildfastwithai.com/blogs/best-ai-models-may-2026-leaderboard","retrieved_at":"2026-05-04T07:13:00Z","title":"Best AI Models: April + May 2026 Leaderboard","excerpt":"GPT-5.5 leads on SWE-bench Verified at 88.7% (vs Claude Opus 4.7's 87.6%). Also scored 82.7% on Terminal-Bench 2.0."},"benchmark_name":{"url":"https://www.buildfastwithai.com/blogs/best-ai-models-may-2026-leaderboard","retrieved_at":"2026-05-04T07:13:00Z","title":"Best AI Models: April + May 2026 Leaderboard","excerpt":"GPT-5.5 SWE-bench Verified 88.7%. Terminal-Bench 2.0 82.7%."},"announcement_url":{"url":"https://openai.com/index/introducing-gpt-5-5/","retrieved_at":"2026-04-30","title":"Introducing GPT-5.5 | OpenAI","excerpt":"We're releasing GPT-5.5, our smartest and most intuitive to use model yet. Terminal-Bench 2.0: 82.7%. Context window: 1M tokens."}}},{"entity_id":"openai-gpt-5-5-pro","fields":{"name":"GPT-5.5 Pro","vendor":"OpenAI","released_at":"2026-04-26","modality":"text, code, vision","access":"closed","context_window":409600,"announcement_url":"https://help.openai.com/en/articles/11909943-gpt-53-and-gpt-55-in-chatgpt"},"evidence":{"name":{"url":"https://help.openai.com/en/articles/11909943-gpt-53-and-gpt-55-in-chatgpt","retrieved_at":"2026-05-03T07:00:00Z","title":"GPT-5.3 and GPT-5.5 in ChatGPT | OpenAI Help Center","excerpt":"GPT-5.5 Pro is the highest-capability GPT-5.5 option in ChatGPT for the hardest tasks and long-running workflows. Pro tier: 400k (272k input + 128k max output)."},"vendor":{"url":"https://help.openai.com/en/articles/11909943-gpt-53-and-gpt-55-in-chatgpt","retrieved_at":"2026-05-03T07:00:00Z","title":"GPT-5.3 and GPT-5.5 in ChatGPT | OpenAI Help Center","excerpt":"GPT-5.5 Pro is the highest-capability GPT-5.5 option in ChatGPT for the hardest tasks and long-running workflows. Pro tier: 400k (272k input + 128k max output)."},"released_at":{"url":"https://help.openai.com/en/articles/11909943-gpt-53-and-gpt-55-in-chatgpt","retrieved_at":"2026-05-03T07:00:00Z","title":"GPT-5.3 and GPT-5.5 in ChatGPT | OpenAI Help Center","excerpt":"GPT-5.5 Pro is the highest-capability GPT-5.5 option in ChatGPT for the hardest tasks and long-running workflows. Pro tier: 400k (272k input + 128k max output)."},"modality":{"url":"https://help.openai.com/en/articles/11909943-gpt-53-and-gpt-55-in-chatgpt","retrieved_at":"2026-05-03T07:00:00Z","title":"GPT-5.3 and GPT-5.5 in ChatGPT | OpenAI Help Center","excerpt":"GPT-5.5 Pro is the highest-capability GPT-5.5 option in ChatGPT for the hardest tasks and long-running workflows. Pro tier: 400k (272k input + 128k max output)."},"access":{"url":"https://help.openai.com/en/articles/11909943-gpt-53-and-gpt-55-in-chatgpt","retrieved_at":"2026-05-03T07:00:00Z","title":"GPT-5.3 and GPT-5.5 in ChatGPT | OpenAI Help Center","excerpt":"GPT-5.5 Pro is the highest-capability GPT-5.5 option in ChatGPT for the hardest tasks and long-running workflows. Pro tier: 400k (272k input + 128k max output)."},"context_window":{"url":"https://help.openai.com/en/articles/11909943-gpt-53-and-gpt-55-in-chatgpt","retrieved_at":"2026-05-03T07:00:00Z","title":"GPT-5.3 and GPT-5.5 in ChatGPT | OpenAI Help Center","excerpt":"GPT-5.5 Pro is the highest-capability GPT-5.5 option in ChatGPT for the hardest tasks and long-running workflows. Pro tier: 400k (272k input + 128k max output)."},"announcement_url":{"url":"https://help.openai.com/en/articles/11909943-gpt-53-and-gpt-55-in-chatgpt","retrieved_at":"2026-05-03T07:00:00Z","title":"GPT-5.3 and GPT-5.5 in ChatGPT | OpenAI Help Center","excerpt":"GPT-5.5 Pro is the highest-capability GPT-5.5 option in ChatGPT for the hardest tasks and long-running workflows. Pro tier: 400k (272k input + 128k max output)."}}},{"entity_id":"openai-gpt-5-5-thinking","fields":{"name":"GPT-5.5 Thinking","vendor":"OpenAI","released_at":"2026-04-26","modality":"text, code, vision","access":"closed","context_window":262144,"announcement_url":"https://help.openai.com/en/articles/11909943-gpt-53-and-gpt-55-in-chatgpt"},"evidence":{"name":{"url":"https://help.openai.com/en/articles/11909943-gpt-53-and-gpt-55-in-chatgpt","retrieved_at":"2026-05-03T07:00:00Z","title":"GPT-5.3 and GPT-5.5 in ChatGPT | OpenAI Help Center","excerpt":"GPT-5.5 Thinking is our most capable reasoning model in ChatGPT. Context: Pro tier 400k (272k input + 128k max output); all paid tiers 256K."},"vendor":{"url":"https://help.openai.com/en/articles/11909943-gpt-53-and-gpt-55-in-chatgpt","retrieved_at":"2026-05-03T07:00:00Z","title":"GPT-5.3 and GPT-5.5 in ChatGPT | OpenAI Help Center","excerpt":"GPT-5.5 Thinking is our most capable reasoning model in ChatGPT. Context: Pro tier 400k (272k input + 128k max output); all paid tiers 256K."},"released_at":{"url":"https://help.openai.com/en/articles/11909943-gpt-53-and-gpt-55-in-chatgpt","retrieved_at":"2026-05-03T07:00:00Z","title":"GPT-5.3 and GPT-5.5 in ChatGPT | OpenAI Help Center","excerpt":"GPT-5.5 Thinking is our most capable reasoning model in ChatGPT. Context: Pro tier 400k (272k input + 128k max output); all paid tiers 256K."},"modality":{"url":"https://help.openai.com/en/articles/11909943-gpt-53-and-gpt-55-in-chatgpt","retrieved_at":"2026-05-03T07:00:00Z","title":"GPT-5.3 and GPT-5.5 in ChatGPT | OpenAI Help Center","excerpt":"GPT-5.5 Thinking is our most capable reasoning model in ChatGPT. Context: Pro tier 400k (272k input + 128k max output); all paid tiers 256K."},"access":{"url":"https://help.openai.com/en/articles/11909943-gpt-53-and-gpt-55-in-chatgpt","retrieved_at":"2026-05-03T07:00:00Z","title":"GPT-5.3 and GPT-5.5 in ChatGPT | OpenAI Help Center","excerpt":"GPT-5.5 Thinking is our most capable reasoning model in ChatGPT. Context: Pro tier 400k (272k input + 128k max output); all paid tiers 256K."},"context_window":{"url":"https://help.openai.com/en/articles/11909943-gpt-53-and-gpt-55-in-chatgpt","retrieved_at":"2026-05-03T07:00:00Z","title":"GPT-5.3 and GPT-5.5 in ChatGPT | OpenAI Help Center","excerpt":"GPT-5.5 Thinking is our most capable reasoning model in ChatGPT. Context: Pro tier 400k (272k input + 128k max output); all paid tiers 256K."},"announcement_url":{"url":"https://help.openai.com/en/articles/11909943-gpt-53-and-gpt-55-in-chatgpt","retrieved_at":"2026-05-03T07:00:00Z","title":"GPT-5.3 and GPT-5.5 in ChatGPT | OpenAI Help Center","excerpt":"GPT-5.5 Thinking is our most capable reasoning model in ChatGPT. Context: Pro tier 400k (272k input + 128k max output); all paid tiers 256K."}}},{"entity_id":"openai-gpt-image-2","fields":{"name":"gpt-image-2 (ChatGPT Images 2.0)","vendor":"OpenAI","released_at":"2026-04-21","modality":"image","access":"closed","announcement_url":"https://openai.com/index/introducing-chatgpt-images-2-0/"},"evidence":{"name":{"url":"https://openai.com/index/introducing-chatgpt-images-2-0/","retrieved_at":"2026-04-30","title":"Introducing ChatGPT Images 2.0 | OpenAI","excerpt":"Introducing ChatGPT Images 2.0: a new era of image generation. Available in ChatGPT and via API as gpt-image-2."},"vendor":{"url":"https://openai.com/index/introducing-chatgpt-images-2-0/","retrieved_at":"2026-04-30","title":"Introducing ChatGPT Images 2.0 | OpenAI","excerpt":"Introducing ChatGPT Images 2.0: a new era of image generation. Available in ChatGPT and via API as gpt-image-2."},"released_at":{"url":"https://openai.com/index/introducing-chatgpt-images-2-0/","retrieved_at":"2026-04-30","title":"Introducing ChatGPT Images 2.0 | OpenAI","excerpt":"Introducing ChatGPT Images 2.0: a new era of image generation. Available in ChatGPT and via API as gpt-image-2."},"modality":{"url":"https://openai.com/index/introducing-chatgpt-images-2-0/","retrieved_at":"2026-04-30","title":"Introducing ChatGPT Images 2.0 | OpenAI","excerpt":"Introducing ChatGPT Images 2.0: a new era of image generation. Available in ChatGPT and via API as gpt-image-2."},"access":{"url":"https://openai.com/index/introducing-chatgpt-images-2-0/","retrieved_at":"2026-04-30","title":"Introducing ChatGPT Images 2.0 | OpenAI","excerpt":"Introducing ChatGPT Images 2.0: a new era of image generation. Available in ChatGPT and via API as gpt-image-2."},"announcement_url":{"url":"https://openai.com/index/introducing-chatgpt-images-2-0/","retrieved_at":"2026-04-30","title":"Introducing ChatGPT Images 2.0 | OpenAI","excerpt":"Introducing ChatGPT Images 2.0: a new era of image generation. Available in ChatGPT and via API as gpt-image-2."}}},{"entity_id":"openai-gpt-rosalind","fields":{"name":"GPT-Rosalind","vendor":"OpenAI","released_at":"2026-04-16","modality":"text, code, biology","access":"closed","announcement_url":"https://openai.com/index/introducing-gpt-rosalind/","benchmark_score":75.1,"benchmark_name":"BixBench Pass@1"},"evidence":{"name":{"url":"https://openai.com/index/introducing-gpt-rosalind/","retrieved_at":"2026-04-30","title":"Introducing GPT-Rosalind for life sciences research | OpenAI","excerpt":"GPT-Rosalind: frontier reasoning model built to support research across biology, drug discovery, and translational medicine. Research preview via trusted access program."},"vendor":{"url":"https://openai.com/index/introducing-gpt-rosalind/","retrieved_at":"2026-04-30","title":"Introducing GPT-Rosalind for life sciences research | OpenAI","excerpt":"GPT-Rosalind: frontier reasoning model built to support research across biology, drug discovery, and translational medicine. Research preview via trusted access program."},"released_at":{"url":"https://openai.com/index/introducing-gpt-rosalind/","retrieved_at":"2026-04-30","title":"Introducing GPT-Rosalind for life sciences research | OpenAI","excerpt":"GPT-Rosalind: frontier reasoning model built to support research across biology, drug discovery, and translational medicine. Research preview via trusted access program."},"modality":{"url":"https://openai.com/index/introducing-gpt-rosalind/","retrieved_at":"2026-05-05T06:56:00Z","title":"Introducing GPT-Rosalind","excerpt":"GPT-Rosalind is a reasoning model tuned specifically for biological research, drug discovery, and translational medicine."},"access":{"url":"https://openai.com/index/introducing-gpt-rosalind/","retrieved_at":"2026-04-30","title":"Introducing GPT-Rosalind for life sciences research | OpenAI","excerpt":"GPT-Rosalind: frontier reasoning model built to support research across biology, drug discovery, and translational medicine. Research preview via trusted access program."},"announcement_url":{"url":"https://openai.com/index/introducing-gpt-rosalind/","retrieved_at":"2026-04-30","title":"Introducing GPT-Rosalind for life sciences research | OpenAI","excerpt":"GPT-Rosalind: frontier reasoning model built to support research across biology, drug discovery, and translational medicine. Research preview via trusted access program."},"benchmark_score":{"url":"https://www.technology.org/2026/04/17/openais-gpt-rosalind-wants-to-shave-years-off-drug-discovery/","retrieved_at":"2026-05-05T06:56:00Z","title":"OpenAI GPT-Rosalind: AI Model for Biology Research","excerpt":"On the BixBench bioinformatics benchmark, the model scored 0.751 Pass@1, ahead of GPT-5.4 (0.732), Grok 4.2 (0.698), GPT-5.2 (0.611)."},"benchmark_name":{"url":"https://www.technology.org/2026/04/17/openais-gpt-rosalind-wants-to-shave-years-off-drug-discovery/","retrieved_at":"2026-05-05T06:56:00Z","title":"OpenAI GPT-Rosalind: AI Model for Biology Research","excerpt":"On the BixBench bioinformatics benchmark, the model scored 0.751 Pass@1"}}},{"entity_id":"qwen-qwen3-5-omni-plus","fields":{"name":"Qwen3.5-Omni-Plus","vendor":"Alibaba Qwen","released_at":"2026-03-30","modality":"text, image, audio, video","access":"closed","parameters":"~30B total / ~3B active","context_window":256000,"benchmark_score":82.2,"benchmark_name":"MMAU","announcement_url":"https://qwen.ai/blog?id=qwen3.5-omni"},"evidence":{"name":{"url":"https://awesomeagents.ai/models/qwen-3-5-omni/","retrieved_at":"2026-05-04T07:12:00Z","title":"Qwen3.5-Omni | Awesome Agents","excerpt":"Qwen3.5-Omni-Plus released March 30, 2026. Thinker-Talker architecture with Hybrid-Attention MoE. Takes text, image, audio, video as input and produces text plus streaming speech. 256K context, 113-language speech recognition."},"vendor":{"url":"https://awesomeagents.ai/models/qwen-3-5-omni/","retrieved_at":"2026-05-04T07:12:00Z","title":"Qwen3.5-Omni | Awesome Agents","excerpt":"Qwen3.5-Omni-Plus released March 30, 2026. Thinker-Talker architecture with Hybrid-Attention MoE. Takes text, image, audio, video as input and produces text plus streaming speech. 256K context, 113-language speech recognition."},"released_at":{"url":"https://awesomeagents.ai/models/qwen-3-5-omni/","retrieved_at":"2026-05-04T07:12:00Z","title":"Qwen3.5-Omni | Awesome Agents","excerpt":"Qwen3.5-Omni-Plus released March 30, 2026. Thinker-Talker architecture with Hybrid-Attention MoE. Takes text, image, audio, video as input and produces text plus streaming speech. 256K context, 113-language speech recognition."},"modality":{"url":"https://awesomeagents.ai/models/qwen-3-5-omni/","retrieved_at":"2026-05-04T07:12:00Z","title":"Qwen3.5-Omni | Awesome Agents","excerpt":"Qwen3.5-Omni-Plus released March 30, 2026. Thinker-Talker architecture with Hybrid-Attention MoE. Takes text, image, audio, video as input and produces text plus streaming speech. 256K context, 113-language speech recognition."},"access":{"url":"https://awesomeagents.ai/models/qwen-3-5-omni/","retrieved_at":"2026-05-04T07:12:00Z","title":"Qwen3.5-Omni | Awesome Agents","excerpt":"Qwen3.5-Omni-Plus released March 30, 2026. Thinker-Talker architecture with Hybrid-Attention MoE. Takes text, image, audio, video as input and produces text plus streaming speech. 256K context, 113-language speech recognition."},"parameters":{"url":"https://awesomeagents.ai/models/qwen-3-5-omni/","retrieved_at":"2026-05-04T07:12:00Z","title":"Qwen3.5-Omni | Awesome Agents","excerpt":"Qwen3.5-Omni-Plus released March 30, 2026. Thinker-Talker architecture with Hybrid-Attention MoE. Takes text, image, audio, video as input and produces text plus streaming speech. 256K context, 113-language speech recognition."},"context_window":{"url":"https://awesomeagents.ai/models/qwen-3-5-omni/","retrieved_at":"2026-05-04T07:12:00Z","title":"Qwen3.5-Omni | Awesome Agents","excerpt":"Qwen3.5-Omni-Plus released March 30, 2026. Thinker-Talker architecture with Hybrid-Attention MoE. Takes text, image, audio, video as input and produces text plus streaming speech. 256K context, 113-language speech recognition."},"benchmark_score":{"url":"https://awesomeagents.ai/models/qwen-3-5-omni/","retrieved_at":"2026-05-04T07:12:00Z","title":"Qwen3.5-Omni | Awesome Agents","excerpt":"Qwen3.5-Omni-Plus released March 30, 2026. Thinker-Talker architecture with Hybrid-Attention MoE. Takes text, image, audio, video as input and produces text plus streaming speech. 256K context, 113-language speech recognition."},"benchmark_name":{"url":"https://awesomeagents.ai/models/qwen-3-5-omni/","retrieved_at":"2026-05-04T07:12:00Z","title":"Qwen3.5-Omni | Awesome Agents","excerpt":"Qwen3.5-Omni-Plus released March 30, 2026. Thinker-Talker architecture with Hybrid-Attention MoE. Takes text, image, audio, video as input and produces text plus streaming speech. 256K context, 113-language speech recognition."},"announcement_url":{"url":"https://awesomeagents.ai/models/qwen-3-5-omni/","retrieved_at":"2026-05-04T07:12:00Z","title":"Qwen3.5-Omni | Awesome Agents","excerpt":"Qwen3.5-Omni-Plus released March 30, 2026. Thinker-Talker architecture with Hybrid-Attention MoE. Takes text, image, audio, video as input and produces text plus streaming speech. 256K context, 113-language speech recognition."}}},{"entity_id":"qwen-qwen3-6-27b","fields":{"name":"Qwen3.6-27B","vendor":"Alibaba Qwen","released_at":"2026-04-22","modality":"text, code, image, video","access":"open-weights","parameters":"27B","announcement_url":"https://qwen.ai/blog?id=qwen3.6-27b","benchmark_score":77.2,"benchmark_name":"SWE-Bench Verified","context_window":262144},"evidence":{"name":{"url":"https://www.marktechpost.com/2026/04/22/alibaba-qwen-team-releases-qwen3-6-27b-a-dense-open-weight-model-outperforming-397b-moe-on-agentic-coding-benchmarks/","retrieved_at":"2026-04-30","title":"Alibaba Qwen Team Releases Qwen3.6-27B - MarkTechPost","excerpt":"Qwen3.6-27B: a dense open-weight model outperforming 397B MoE on agentic coding benchmarks. Released April 22, 2026."},"vendor":{"url":"https://www.marktechpost.com/2026/04/22/alibaba-qwen-team-releases-qwen3-6-27b-a-dense-open-weight-model-outperforming-397b-moe-on-agentic-coding-benchmarks/","retrieved_at":"2026-04-30","title":"Alibaba Qwen Team Releases Qwen3.6-27B - MarkTechPost","excerpt":"Qwen3.6-27B: a dense open-weight model outperforming 397B MoE on agentic coding benchmarks. Released April 22, 2026."},"released_at":{"url":"https://www.marktechpost.com/2026/04/22/alibaba-qwen-team-releases-qwen3-6-27b-a-dense-open-weight-model-outperforming-397b-moe-on-agentic-coding-benchmarks/","retrieved_at":"2026-04-30","title":"Alibaba Qwen Team Releases Qwen3.6-27B - MarkTechPost","excerpt":"Qwen3.6-27B: a dense open-weight model outperforming 397B MoE on agentic coding benchmarks. Released April 22, 2026."},"modality":{"url":"https://www.marktechpost.com/2026/04/22/alibaba-qwen-team-releases-qwen3-6-27b-a-dense-open-weight-model-outperforming-397b-moe-on-agentic-coding-benchmarks/","retrieved_at":"2026-05-01T07:00:00Z","title":"Alibaba Qwen Team Releases Qwen3.6-27B","excerpt":"Qwen3.6-27B scores 77.2 on SWE-bench Verified, 59.3 on Terminal-Bench 2.0, 83.9 on LiveCodeBench v6. Native context 262,144 tokens (1M with YaRN). Natively multimodal: text, image, video."},"access":{"url":"https://www.marktechpost.com/2026/04/22/alibaba-qwen-team-releases-qwen3-6-27b-a-dense-open-weight-model-outperforming-397b-moe-on-agentic-coding-benchmarks/","retrieved_at":"2026-04-30","title":"Alibaba Qwen Team Releases Qwen3.6-27B - MarkTechPost","excerpt":"Qwen3.6-27B: a dense open-weight model outperforming 397B MoE on agentic coding benchmarks. Released April 22, 2026."},"parameters":{"url":"https://www.marktechpost.com/2026/04/22/alibaba-qwen-team-releases-qwen3-6-27b-a-dense-open-weight-model-outperforming-397b-moe-on-agentic-coding-benchmarks/","retrieved_at":"2026-04-30","title":"Alibaba Qwen Team Releases Qwen3.6-27B - MarkTechPost","excerpt":"Qwen3.6-27B: a dense open-weight model outperforming 397B MoE on agentic coding benchmarks. Released April 22, 2026."},"announcement_url":{"url":"https://qwen.ai/blog?id=qwen3.6-27b","retrieved_at":"2026-05-01T07:00:00Z","title":"Qwen3.6-27B official blog","excerpt":"Official Qwen blog post for Qwen3.6-27B at qwen.ai"},"benchmark_score":{"url":"https://www.marktechpost.com/2026/04/22/alibaba-qwen-team-releases-qwen3-6-27b-a-dense-open-weight-model-outperforming-397b-moe-on-agentic-coding-benchmarks/","retrieved_at":"2026-05-01T07:00:00Z","title":"Alibaba Qwen Team Releases Qwen3.6-27B","excerpt":"Qwen3.6-27B scores 77.2 on SWE-bench Verified, 59.3 on Terminal-Bench 2.0, 83.9 on LiveCodeBench v6. Native context 262,144 tokens (1M with YaRN). Natively multimodal: text, image, video."},"benchmark_name":{"url":"https://www.marktechpost.com/2026/04/22/alibaba-qwen-team-releases-qwen3-6-27b-a-dense-open-weight-model-outperforming-397b-moe-on-agentic-coding-benchmarks/","retrieved_at":"2026-05-01T07:00:00Z","title":"Alibaba Qwen Team Releases Qwen3.6-27B","excerpt":"Qwen3.6-27B scores 77.2 on SWE-bench Verified, 59.3 on Terminal-Bench 2.0, 83.9 on LiveCodeBench v6. Native context 262,144 tokens (1M with YaRN). Natively multimodal: text, image, video."},"context_window":{"url":"https://www.marktechpost.com/2026/04/22/alibaba-qwen-team-releases-qwen3-6-27b-a-dense-open-weight-model-outperforming-397b-moe-on-agentic-coding-benchmarks/","retrieved_at":"2026-05-01T07:00:00Z","title":"Alibaba Qwen Team Releases Qwen3.6-27B","excerpt":"Qwen3.6-27B scores 77.2 on SWE-bench Verified, 59.3 on Terminal-Bench 2.0, 83.9 on LiveCodeBench v6. Native context 262,144 tokens (1M with YaRN). Natively multimodal: text, image, video."}}},{"entity_id":"qwen-qwen3-6-35b-a3b","fields":{"name":"Qwen3.6-35B-A3B","vendor":"Alibaba Qwen","released_at":"2026-04-14","modality":"text, code","access":"open-weights","parameters":"35B-A3B","announcement_url":"https://qwen.ai/blog?id=qwen3.6-35b-a3b"},"evidence":{"name":{"url":"https://qwen.ai/blog?id=qwen3.6-35b-a3b","retrieved_at":"2026-04-30","title":"Qwen3.6-35B-A3B: Agentic Coding Power, Now Open to All","excerpt":"Qwen3.6-35B-A3B: Agentic Coding Power, Now Open to All. MoE model released April 14, 2026."},"vendor":{"url":"https://qwen.ai/blog?id=qwen3.6-35b-a3b","retrieved_at":"2026-04-30","title":"Qwen3.6-35B-A3B: Agentic Coding Power, Now Open to All","excerpt":"Qwen3.6-35B-A3B: Agentic Coding Power, Now Open to All. MoE model released April 14, 2026."},"released_at":{"url":"https://qwen.ai/blog?id=qwen3.6-35b-a3b","retrieved_at":"2026-04-30","title":"Qwen3.6-35B-A3B: Agentic Coding Power, Now Open to All","excerpt":"Qwen3.6-35B-A3B: Agentic Coding Power, Now Open to All. MoE model released April 14, 2026."},"modality":{"url":"https://qwen.ai/blog?id=qwen3.6-35b-a3b","retrieved_at":"2026-04-30","title":"Qwen3.6-35B-A3B: Agentic Coding Power, Now Open to All","excerpt":"Qwen3.6-35B-A3B: Agentic Coding Power, Now Open to All. MoE model released April 14, 2026."},"access":{"url":"https://qwen.ai/blog?id=qwen3.6-35b-a3b","retrieved_at":"2026-04-30","title":"Qwen3.6-35B-A3B: Agentic Coding Power, Now Open to All","excerpt":"Qwen3.6-35B-A3B: Agentic Coding Power, Now Open to All. MoE model released April 14, 2026."},"parameters":{"url":"https://qwen.ai/blog?id=qwen3.6-35b-a3b","retrieved_at":"2026-04-30","title":"Qwen3.6-35B-A3B: Agentic Coding Power, Now Open to All","excerpt":"Qwen3.6-35B-A3B: Agentic Coding Power, Now Open to All. MoE model released April 14, 2026."},"announcement_url":{"url":"https://qwen.ai/blog?id=qwen3.6-35b-a3b","retrieved_at":"2026-04-30","title":"Qwen3.6-35B-A3B: Agentic Coding Power, Now Open to All","excerpt":"Qwen3.6-35B-A3B: Agentic Coding Power, Now Open to All. MoE model released April 14, 2026."}}},{"entity_id":"xai-grok-4-3","fields":{"name":"Grok 4.3","vendor":"xAI","released_at":"2026-05-01","modality":"text, code, vision","access":"closed","benchmark_score":53,"benchmark_name":"Artificial Analysis Intelligence Index","announcement_url":"https://x.ai/news","context_window":1000000},"evidence":{"name":{"url":"https://venturebeat.com/technology/xai-launches-grok-4-3-at-an-aggressively-low-price-and-a-new-fast-powerful-voice-cloning-suite","retrieved_at":"2026-05-03T07:00:00Z","title":"xAI launches Grok 4.3 at an aggressively low price and a new, fast, powerful voice cloning suite | VentureBeat","excerpt":"xAI launches Grok 4.3 at an aggressively low price. Grok 4.3 ranks #1 on CaseLaw v2 (79.3% accuracy) and #1 on CorpFin. Artificial Analysis corroborated this performance, scoring an Elo of 1500 on the GDPval-AA benchmark. Scores 53 on the Artificial Analysis Intelligence Index."},"vendor":{"url":"https://venturebeat.com/technology/xai-launches-grok-4-3-at-an-aggressively-low-price-and-a-new-fast-powerful-voice-cloning-suite","retrieved_at":"2026-05-03T07:00:00Z","title":"xAI launches Grok 4.3 at an aggressively low price and a new, fast, powerful voice cloning suite | VentureBeat","excerpt":"xAI launches Grok 4.3 at an aggressively low price. Grok 4.3 ranks #1 on CaseLaw v2 (79.3% accuracy) and #1 on CorpFin. Artificial Analysis corroborated this performance, scoring an Elo of 1500 on the GDPval-AA benchmark. Scores 53 on the Artificial Analysis Intelligence Index."},"released_at":{"url":"https://venturebeat.com/technology/xai-launches-grok-4-3-at-an-aggressively-low-price-and-a-new-fast-powerful-voice-cloning-suite","retrieved_at":"2026-05-03T07:00:00Z","title":"xAI launches Grok 4.3 at an aggressively low price and a new, fast, powerful voice cloning suite | VentureBeat","excerpt":"xAI launches Grok 4.3 at an aggressively low price. Grok 4.3 ranks #1 on CaseLaw v2 (79.3% accuracy) and #1 on CorpFin. Artificial Analysis corroborated this performance, scoring an Elo of 1500 on the GDPval-AA benchmark. Scores 53 on the Artificial Analysis Intelligence Index."},"modality":{"url":"https://artificialanalysis.ai/models/grok-4-3","retrieved_at":"2026-05-04T07:14:00Z","title":"Grok 4.3 - Intelligence, Performance & Price Analysis | Artificial Analysis","excerpt":"Grok 4.3 supports text and image input, and text output. Yes, Grok 4.3 is multimodal."},"access":{"url":"https://venturebeat.com/technology/xai-launches-grok-4-3-at-an-aggressively-low-price-and-a-new-fast-powerful-voice-cloning-suite","retrieved_at":"2026-05-03T07:00:00Z","title":"xAI launches Grok 4.3 at an aggressively low price and a new, fast, powerful voice cloning suite | VentureBeat","excerpt":"xAI launches Grok 4.3 at an aggressively low price. Grok 4.3 ranks #1 on CaseLaw v2 (79.3% accuracy) and #1 on CorpFin. Artificial Analysis corroborated this performance, scoring an Elo of 1500 on the GDPval-AA benchmark. Scores 53 on the Artificial Analysis Intelligence Index."},"benchmark_score":{"url":"https://venturebeat.com/technology/xai-launches-grok-4-3-at-an-aggressively-low-price-and-a-new-fast-powerful-voice-cloning-suite","retrieved_at":"2026-05-03T07:00:00Z","title":"xAI launches Grok 4.3 at an aggressively low price and a new, fast, powerful voice cloning suite | VentureBeat","excerpt":"xAI launches Grok 4.3 at an aggressively low price. Grok 4.3 ranks #1 on CaseLaw v2 (79.3% accuracy) and #1 on CorpFin. Artificial Analysis corroborated this performance, scoring an Elo of 1500 on the GDPval-AA benchmark. Scores 53 on the Artificial Analysis Intelligence Index."},"benchmark_name":{"url":"https://venturebeat.com/technology/xai-launches-grok-4-3-at-an-aggressively-low-price-and-a-new-fast-powerful-voice-cloning-suite","retrieved_at":"2026-05-03T07:00:00Z","title":"xAI launches Grok 4.3 at an aggressively low price and a new, fast, powerful voice cloning suite | VentureBeat","excerpt":"xAI launches Grok 4.3 at an aggressively low price. Grok 4.3 ranks #1 on CaseLaw v2 (79.3% accuracy) and #1 on CorpFin. Artificial Analysis corroborated this performance, scoring an Elo of 1500 on the GDPval-AA benchmark. Scores 53 on the Artificial Analysis Intelligence Index."},"announcement_url":{"url":"https://venturebeat.com/technology/xai-launches-grok-4-3-at-an-aggressively-low-price-and-a-new-fast-powerful-voice-cloning-suite","retrieved_at":"2026-05-03T07:00:00Z","title":"xAI launches Grok 4.3 at an aggressively low price and a new, fast, powerful voice cloning suite | VentureBeat","excerpt":"xAI launches Grok 4.3 at an aggressively low price. Grok 4.3 ranks #1 on CaseLaw v2 (79.3% accuracy) and #1 on CorpFin. Artificial Analysis corroborated this performance, scoring an Elo of 1500 on the GDPval-AA benchmark. Scores 53 on the Artificial Analysis Intelligence Index."},"context_window":{"url":"https://artificialanalysis.ai/models/grok-4-3","retrieved_at":"2026-05-04T07:14:00Z","title":"Grok 4.3 - Intelligence, Performance & Price Analysis | Artificial Analysis","excerpt":"Grok 4.3 has a context window of 1.0M tokens. Grok 4.3 supports text and image input."}}},{"entity_id":"xai-grok-voice-think-fast-1","fields":{"name":"Grok Voice Think Fast 1.0","vendor":"xAI","released_at":"2026-04-23","modality":"audio","access":"closed","announcement_url":"https://x.ai/news/grok-voice-think-fast-1","benchmark_name":"τ-voice Bench"},"evidence":{"name":{"url":"https://x.ai/news/grok-voice-think-fast-1","retrieved_at":"2026-05-02T06:48:41Z","title":"Grok Voice Think Fast 1.0 | xAI","excerpt":"Today, we're excited to announce a step change in xAI's Voice Agent capabilities: Introducing grok-voice-think-fast-1.0 — our new flagship voice model. This new model excels at complex, ambiguous, multi-step workflows across customer support, sales, and enterprise applications."},"vendor":{"url":"https://x.ai/news/grok-voice-think-fast-1","retrieved_at":"2026-05-02T06:48:41Z","title":"Grok Voice Think Fast 1.0 | xAI","excerpt":"Today, we're excited to announce a step change in xAI's Voice Agent capabilities: Introducing grok-voice-think-fast-1.0 — our new flagship voice model. This new model excels at complex, ambiguous, multi-step workflows across customer support, sales, and enterprise applications."},"released_at":{"url":"https://x.ai/news/grok-voice-think-fast-1","retrieved_at":"2026-05-02T06:48:41Z","title":"Grok Voice Think Fast 1.0 | xAI","excerpt":"Today, we're excited to announce a step change in xAI's Voice Agent capabilities: Introducing grok-voice-think-fast-1.0 — our new flagship voice model. This new model excels at complex, ambiguous, multi-step workflows across customer support, sales, and enterprise applications."},"modality":{"url":"https://x.ai/news/grok-voice-think-fast-1","retrieved_at":"2026-05-02T06:48:41Z","title":"Grok Voice Think Fast 1.0 | xAI","excerpt":"Today, we're excited to announce a step change in xAI's Voice Agent capabilities: Introducing grok-voice-think-fast-1.0 — our new flagship voice model. This new model excels at complex, ambiguous, multi-step workflows across customer support, sales, and enterprise applications."},"access":{"url":"https://x.ai/news/grok-voice-think-fast-1","retrieved_at":"2026-05-02T06:48:41Z","title":"Grok Voice Think Fast 1.0 | xAI","excerpt":"Today, we're excited to announce a step change in xAI's Voice Agent capabilities: Introducing grok-voice-think-fast-1.0 — our new flagship voice model. This new model excels at complex, ambiguous, multi-step workflows across customer support, sales, and enterprise applications."},"announcement_url":{"url":"https://x.ai/news/grok-voice-think-fast-1","retrieved_at":"2026-05-02T06:48:41Z","title":"Grok Voice Think Fast 1.0 | xAI","excerpt":"Today, we're excited to announce a step change in xAI's Voice Agent capabilities: Introducing grok-voice-think-fast-1.0 — our new flagship voice model. This new model excels at complex, ambiguous, multi-step workflows across customer support, sales, and enterprise applications."},"benchmark_name":{"url":"https://x.ai/news/grok-voice-think-fast-1","retrieved_at":"2026-05-02T06:48:41Z","title":"Grok Voice Think Fast 1.0 | xAI","excerpt":"Today, we're excited to announce a step change in xAI's Voice Agent capabilities: Introducing grok-voice-think-fast-1.0 — our new flagship voice model. This new model excels at complex, ambiguous, multi-step workflows across customer support, sales, and enterprise applications."}}},{"entity_id":"xiaomi-mimo-v2-5","fields":{"name":"MiMo-V2.5","vendor":"Xiaomi","released_at":"2026-04-27","modality":"text, image, video, audio, code","access":"open-weights","parameters":"310B total / 15B active","context_window":1000000,"benchmark_score":65.8,"benchmark_name":"Terminal-Bench 2.0","announcement_url":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5"},"evidence":{"name":{"url":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5","retrieved_at":"2026-05-03T07:00:00Z","title":"XiaomiMiMo/MiMo-V2.5 - Hugging Face","excerpt":"MiMo-V2.5 is a native omnimodal model with strong agentic capabilities. Architecture: Sparse MoE, 310B total / 15B activated parameters. Context Length: Up to 1M tokens. Modalities: Text, Image, Video, Audio. 21,407 downloads. Terminal-Bench 2.0: 65.8, SWE Bench Pro: 56.1."},"vendor":{"url":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5","retrieved_at":"2026-05-03T07:00:00Z","title":"XiaomiMiMo/MiMo-V2.5 - Hugging Face","excerpt":"MiMo-V2.5 is a native omnimodal model with strong agentic capabilities. Architecture: Sparse MoE, 310B total / 15B activated parameters. Context Length: Up to 1M tokens. Modalities: Text, Image, Video, Audio. 21,407 downloads. Terminal-Bench 2.0: 65.8, SWE Bench Pro: 56.1."},"released_at":{"url":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5","retrieved_at":"2026-05-03T07:00:00Z","title":"XiaomiMiMo/MiMo-V2.5 - Hugging Face","excerpt":"MiMo-V2.5 is a native omnimodal model with strong agentic capabilities. Architecture: Sparse MoE, 310B total / 15B activated parameters. Context Length: Up to 1M tokens. Modalities: Text, Image, Video, Audio. 21,407 downloads. Terminal-Bench 2.0: 65.8, SWE Bench Pro: 56.1."},"modality":{"url":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5","retrieved_at":"2026-05-03T07:00:00Z","title":"XiaomiMiMo/MiMo-V2.5 - Hugging Face","excerpt":"MiMo-V2.5 is a native omnimodal model with strong agentic capabilities. Architecture: Sparse MoE, 310B total / 15B activated parameters. Context Length: Up to 1M tokens. Modalities: Text, Image, Video, Audio. 21,407 downloads. Terminal-Bench 2.0: 65.8, SWE Bench Pro: 56.1."},"access":{"url":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5","retrieved_at":"2026-05-03T07:00:00Z","title":"XiaomiMiMo/MiMo-V2.5 - Hugging Face","excerpt":"MiMo-V2.5 is a native omnimodal model with strong agentic capabilities. Architecture: Sparse MoE, 310B total / 15B activated parameters. Context Length: Up to 1M tokens. Modalities: Text, Image, Video, Audio. 21,407 downloads. Terminal-Bench 2.0: 65.8, SWE Bench Pro: 56.1."},"parameters":{"url":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5","retrieved_at":"2026-05-03T07:00:00Z","title":"XiaomiMiMo/MiMo-V2.5 - Hugging Face","excerpt":"MiMo-V2.5 is a native omnimodal model with strong agentic capabilities. Architecture: Sparse MoE, 310B total / 15B activated parameters. Context Length: Up to 1M tokens. Modalities: Text, Image, Video, Audio. 21,407 downloads. Terminal-Bench 2.0: 65.8, SWE Bench Pro: 56.1."},"context_window":{"url":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5","retrieved_at":"2026-05-03T07:00:00Z","title":"XiaomiMiMo/MiMo-V2.5 - Hugging Face","excerpt":"MiMo-V2.5 is a native omnimodal model with strong agentic capabilities. Architecture: Sparse MoE, 310B total / 15B activated parameters. Context Length: Up to 1M tokens. Modalities: Text, Image, Video, Audio. 21,407 downloads. Terminal-Bench 2.0: 65.8, SWE Bench Pro: 56.1."},"benchmark_score":{"url":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5","retrieved_at":"2026-05-03T07:00:00Z","title":"XiaomiMiMo/MiMo-V2.5 - Hugging Face","excerpt":"MiMo-V2.5 is a native omnimodal model with strong agentic capabilities. Architecture: Sparse MoE, 310B total / 15B activated parameters. Context Length: Up to 1M tokens. Modalities: Text, Image, Video, Audio. 21,407 downloads. Terminal-Bench 2.0: 65.8, SWE Bench Pro: 56.1."},"benchmark_name":{"url":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5","retrieved_at":"2026-05-03T07:00:00Z","title":"XiaomiMiMo/MiMo-V2.5 - Hugging Face","excerpt":"MiMo-V2.5 is a native omnimodal model with strong agentic capabilities. Architecture: Sparse MoE, 310B total / 15B activated parameters. Context Length: Up to 1M tokens. Modalities: Text, Image, Video, Audio. 21,407 downloads. Terminal-Bench 2.0: 65.8, SWE Bench Pro: 56.1."},"announcement_url":{"url":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5","retrieved_at":"2026-05-03T07:00:00Z","title":"XiaomiMiMo/MiMo-V2.5 - Hugging Face","excerpt":"MiMo-V2.5 is a native omnimodal model with strong agentic capabilities. Architecture: Sparse MoE, 310B total / 15B activated parameters. Context Length: Up to 1M tokens. Modalities: Text, Image, Video, Audio. 21,407 downloads. Terminal-Bench 2.0: 65.8, SWE Bench Pro: 56.1."}}},{"entity_id":"xiaomi-mimo-v2-5-pro","fields":{"name":"MiMo-V2.5-Pro","vendor":"Xiaomi","released_at":"2026-04-27","modality":"text, code","access":"open-weights","parameters":"1.02T total / 42B active","context_window":1000000,"benchmark_score":78.9,"benchmark_name":"SWE-Bench Verified","announcement_url":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5-Pro"},"evidence":{"name":{"url":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5-Pro","retrieved_at":"2026-05-04T07:11:00Z","title":"MiMo-V2.5-Pro - Hugging Face","excerpt":"MiMo-V2.5-Pro is an open-source Mixture-of-Experts (MoE) language model with 1.02T total parameters and 42B active parameters. Released 2026-04-27."},"vendor":{"url":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5-Pro","retrieved_at":"2026-05-04T07:11:00Z","title":"MiMo-V2.5-Pro - Hugging Face","excerpt":"MiMo-V2.5-Pro is an open-source Mixture-of-Experts (MoE) language model with 1.02T total parameters and 42B active parameters. Released 2026-04-27."},"released_at":{"url":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5-Pro","retrieved_at":"2026-05-04T07:11:00Z","title":"MiMo-V2.5-Pro - Hugging Face","excerpt":"MiMo-V2.5-Pro is an open-source Mixture-of-Experts (MoE) language model with 1.02T total parameters and 42B active parameters. Released 2026-04-27."},"modality":{"url":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5-Pro","retrieved_at":"2026-05-04T07:11:00Z","title":"MiMo-V2.5-Pro - Hugging Face","excerpt":"MiMo-V2.5-Pro is an open-source Mixture-of-Experts (MoE) language model with 1.02T total parameters and 42B active parameters. Released 2026-04-27."},"access":{"url":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5-Pro","retrieved_at":"2026-05-04T07:11:00Z","title":"MiMo-V2.5-Pro - Hugging Face","excerpt":"MiMo-V2.5-Pro is an open-source Mixture-of-Experts (MoE) language model with 1.02T total parameters and 42B active parameters. Released 2026-04-27."},"parameters":{"url":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5-Pro","retrieved_at":"2026-05-04T07:11:00Z","title":"MiMo-V2.5-Pro - Hugging Face","excerpt":"MiMo-V2.5-Pro is an open-source Mixture-of-Experts (MoE) language model with 1.02T total parameters and 42B active parameters. Released 2026-04-27."},"context_window":{"url":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5-Pro","retrieved_at":"2026-05-04T07:11:00Z","title":"MiMo-V2.5-Pro - Hugging Face","excerpt":"MiMo-V2.5-Pro is an open-source Mixture-of-Experts (MoE) language model with 1.02T total parameters and 42B active parameters. Released 2026-04-27."},"benchmark_score":{"url":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5-Pro","retrieved_at":"2026-05-04T07:11:00Z","title":"MiMo-V2.5-Pro - Hugging Face","excerpt":"MiMo-V2.5-Pro is an open-source Mixture-of-Experts (MoE) language model with 1.02T total parameters and 42B active parameters. Released 2026-04-27."},"benchmark_name":{"url":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5-Pro","retrieved_at":"2026-05-04T07:11:00Z","title":"MiMo-V2.5-Pro - Hugging Face","excerpt":"MiMo-V2.5-Pro is an open-source Mixture-of-Experts (MoE) language model with 1.02T total parameters and 42B active parameters. Released 2026-04-27."},"announcement_url":{"url":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5-Pro","retrieved_at":"2026-05-04T07:11:00Z","title":"MiMo-V2.5-Pro - Hugging Face","excerpt":"MiMo-V2.5-Pro is an open-source Mixture-of-Experts (MoE) language model with 1.02T total parameters and 42B active parameters. Released 2026-04-27."}}},{"entity_id":"zhipuai-glm-5-1","fields":{"name":"GLM-5.1","vendor":"Z.ai / Zhipu AI","released_at":"2026-04-07","modality":"text, code","access":"open-weights","parameters":"744B total / 40B active","benchmark_score":58.4,"benchmark_name":"SWE-Bench Pro","announcement_url":"https://huggingface.co/THUDM/GLM-5.1"},"evidence":{"name":{"url":"https://www.buildfastwithai.com/blogs/best-ai-models-may-2026-leaderboard","retrieved_at":"2026-05-03T07:00:00Z","title":"Best AI Models: April + May 2026 Leaderboard (GPT-5.5, Claude Opus 4.7, DeepSeek V4)","excerpt":"On April 7, 2026, GLM-5.1 became the first open-weight model in history to top the SWE-bench Pro leaderboard, scoring 58.4%. GLM-5.1 is a 744 billion parameter Mixture-of-Experts model with 40 billion active parameters. Trained entirely on 100,000 Huawei Ascend 910B chips."},"vendor":{"url":"https://www.buildfastwithai.com/blogs/best-ai-models-may-2026-leaderboard","retrieved_at":"2026-05-03T07:00:00Z","title":"Best AI Models: April + May 2026 Leaderboard (GPT-5.5, Claude Opus 4.7, DeepSeek V4)","excerpt":"On April 7, 2026, GLM-5.1 became the first open-weight model in history to top the SWE-bench Pro leaderboard, scoring 58.4%. GLM-5.1 is a 744 billion parameter Mixture-of-Experts model with 40 billion active parameters. Trained entirely on 100,000 Huawei Ascend 910B chips."},"released_at":{"url":"https://www.buildfastwithai.com/blogs/best-ai-models-may-2026-leaderboard","retrieved_at":"2026-05-03T07:00:00Z","title":"Best AI Models: April + May 2026 Leaderboard (GPT-5.5, Claude Opus 4.7, DeepSeek V4)","excerpt":"On April 7, 2026, GLM-5.1 became the first open-weight model in history to top the SWE-bench Pro leaderboard, scoring 58.4%. GLM-5.1 is a 744 billion parameter Mixture-of-Experts model with 40 billion active parameters. Trained entirely on 100,000 Huawei Ascend 910B chips."},"modality":{"url":"https://www.buildfastwithai.com/blogs/best-ai-models-may-2026-leaderboard","retrieved_at":"2026-05-03T07:00:00Z","title":"Best AI Models: April + May 2026 Leaderboard (GPT-5.5, Claude Opus 4.7, DeepSeek V4)","excerpt":"On April 7, 2026, GLM-5.1 became the first open-weight model in history to top the SWE-bench Pro leaderboard, scoring 58.4%. GLM-5.1 is a 744 billion parameter Mixture-of-Experts model with 40 billion active parameters. Trained entirely on 100,000 Huawei Ascend 910B chips."},"access":{"url":"https://www.buildfastwithai.com/blogs/best-ai-models-may-2026-leaderboard","retrieved_at":"2026-05-03T07:00:00Z","title":"Best AI Models: April + May 2026 Leaderboard (GPT-5.5, Claude Opus 4.7, DeepSeek V4)","excerpt":"On April 7, 2026, GLM-5.1 became the first open-weight model in history to top the SWE-bench Pro leaderboard, scoring 58.4%. GLM-5.1 is a 744 billion parameter Mixture-of-Experts model with 40 billion active parameters. Trained entirely on 100,000 Huawei Ascend 910B chips."},"parameters":{"url":"https://www.buildfastwithai.com/blogs/best-ai-models-may-2026-leaderboard","retrieved_at":"2026-05-03T07:00:00Z","title":"Best AI Models: April + May 2026 Leaderboard (GPT-5.5, Claude Opus 4.7, DeepSeek V4)","excerpt":"On April 7, 2026, GLM-5.1 became the first open-weight model in history to top the SWE-bench Pro leaderboard, scoring 58.4%. GLM-5.1 is a 744 billion parameter Mixture-of-Experts model with 40 billion active parameters. Trained entirely on 100,000 Huawei Ascend 910B chips."},"benchmark_score":{"url":"https://www.buildfastwithai.com/blogs/best-ai-models-may-2026-leaderboard","retrieved_at":"2026-05-03T07:00:00Z","title":"Best AI Models: April + May 2026 Leaderboard (GPT-5.5, Claude Opus 4.7, DeepSeek V4)","excerpt":"On April 7, 2026, GLM-5.1 became the first open-weight model in history to top the SWE-bench Pro leaderboard, scoring 58.4%. GLM-5.1 is a 744 billion parameter Mixture-of-Experts model with 40 billion active parameters. Trained entirely on 100,000 Huawei Ascend 910B chips."},"benchmark_name":{"url":"https://www.buildfastwithai.com/blogs/best-ai-models-may-2026-leaderboard","retrieved_at":"2026-05-03T07:00:00Z","title":"Best AI Models: April + May 2026 Leaderboard (GPT-5.5, Claude Opus 4.7, DeepSeek V4)","excerpt":"On April 7, 2026, GLM-5.1 became the first open-weight model in history to top the SWE-bench Pro leaderboard, scoring 58.4%. GLM-5.1 is a 744 billion parameter Mixture-of-Experts model with 40 billion active parameters. Trained entirely on 100,000 Huawei Ascend 910B chips."},"announcement_url":{"url":"https://www.buildfastwithai.com/blogs/best-ai-models-may-2026-leaderboard","retrieved_at":"2026-05-03T07:00:00Z","title":"Best AI Models: April + May 2026 Leaderboard (GPT-5.5, Claude Opus 4.7, DeepSeek V4)","excerpt":"On April 7, 2026, GLM-5.1 became the first open-weight model in history to top the SWE-bench Pro leaderboard, scoring 58.4%. GLM-5.1 is a 744 billion parameter Mixture-of-Experts model with 40 billion active parameters. Trained entirely on 100,000 Huawei Ascend 910B chips."}}}],"page":{"limit":50,"next_cursor":null,"has_more":false}}